Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/cmd/scripts/dtest.pl
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/cmd/scripts/dtest.pl	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/cmd/scripts/dtest.pl	(revision 277945)
@@ -1,706 +1,706 @@
 #!/usr/local/bin/perl
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2008 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 
 require 5.8.4;
 
 use File::Find;
 use File::Basename;
 use Getopt::Std;
 use Cwd;
 use Cwd 'abs_path';
 
 $PNAME = $0;
 $PNAME =~ s:.*/::;
 $OPTSTR = 'abd:fghi:jlnqsx:';
 $USAGE = "Usage: $PNAME [-abfghjlnqs] [-d dir] [-i isa] "
     . "[-x opt[=arg]] [file | dir ...]\n";
 ($MACH = `uname -p`) =~ s/\W*\n//;
 ($PLATFORM = `uname -i`) =~ s/\W*\n//;
 
 @dtrace_argv = ();
 
 $ksh_path = '/usr/local/bin/ksh';
 
 @files = ();
 %exceptions = ();
 %results = ();
 $errs = 0;
 
 #
 # If no test files are specified on the command-line, execute a find on "."
 # and append any tst.*.d, tst.*.ksh, err.*.d or drp.*.d files found within
 # the directory tree.
 #
 sub wanted
 {
 	push(@files, $File::Find::name)
 	    if ($_ =~ /^(tst|err|drp)\..+\.(d|ksh)$/ && -f "$_");
 }
 
 sub dirname {
 	my($s) = @_;
 	my($i);
 
 	$s = substr($s, 0, $i) if (($i = rindex($s, '/')) != -1);
 	return $i == -1 ? '.' : $i == 0 ? '/' : $s;
 }
 
 sub usage
 {
 	print $USAGE;
 	print "\t -a  execute test suite using anonymous enablings\n";
 	print "\t -b  execute bad ioctl test program\n";
 	print "\t -d  specify directory for test results files and cores\n";
 	print "\t -g  enable libumem debugging when running tests\n";
 	print "\t -f  force bypassed tests to run\n";
 	print "\t -h  display verbose usage message\n";
 	print "\t -i  specify ISA to test instead of isaexec(3C) default\n";
 	print "\t -j  execute test suite using jdtrace (Java API) only\n";
 	print "\t -l  save log file of results and PIDs used by tests\n";
 	print "\t -n  execute test suite using dtrace(1m) only\n";
 	print "\t -q  set quiet mode (only report errors and summary)\n";
 	print "\t -s  save results files even for tests that pass\n";
 	print "\t -x  pass corresponding -x argument to dtrace(1M)\n";
 	exit(2);
 }
 
 sub errmsg
 {
 	my($msg) = @_;
 
 	print STDERR $msg;
 	print LOG $msg if ($opt_l);
 	$errs++;
 }
 
 sub fail
 {
 	my(@parms) = @_;
 	my($msg) = $parms[0];
 	my($errfile) = $parms[1];
 	my($n) = 0;
 	my($dest) = basename($file);
 
 	while (-d "$opt_d/failure.$n") {
 		$n++;
 	}
 
 	unless (mkdir "$opt_d/failure.$n") {
 		warn "ERROR: failed to make directory $opt_d/failure.$n: $!\n";
 		exit(125);
 	}
 
 	open(README, ">$opt_d/failure.$n/README");
 	print README "ERROR: " . $file . " " . $msg;
 	
 	if (scalar @parms > 1) {
 		print README "; see $errfile\n";
 	} else {
 		if (-f "$opt_d/$pid.core") {
 			print README "; see $pid.core\n";
 		} else {
 			print README "\n";
 		}
 	}
 
 	close(README);
 
 	if (-f "$opt_d/$pid.out") {
 		rename("$opt_d/$pid.out", "$opt_d/failure.$n/$pid.out");
 		link("$file.out", "$opt_d/failure.$n/$dest.out");
 	}
 
 	if (-f "$opt_d/$pid.err") {
 		rename("$opt_d/$pid.err", "$opt_d/failure.$n/$pid.err");
 		link("$file.err", "$opt_d/failure.$n/$dest.err");
 	}
 
 	if (-f "$opt_d/$pid.core") {
 		rename("$opt_d/$pid.core", "$opt_d/failure.$n/$pid.core");
 	}
 
 	link("$file", "$opt_d/failure.$n/$dest");
 
 	$msg = "ERROR: " . $dest . " " . $msg;
 
 	if (scalar @parms > 1) {
 		$msg = $msg . "; see $errfile in failure.$n\n";
 	} else {
 		$msg = $msg . "; details in failure.$n\n";
 	}
 
 	errmsg($msg);
 }
 
 sub logmsg
 {
 	my($msg) = @_;
 
 	print STDOUT $msg unless ($opt_q);
 	print LOG $msg if ($opt_l);
 }
 
 # Trim leading and trailing whitespace
 sub trim {
 	my($s) = @_;
 
 	$s =~ s/^\s*//;
 	$s =~ s/\s*$//;
 	return $s;
 }
 
 # Load exception set of skipped tests from the file at the given
 # pathname. The test names are assumed to be paths relative to $dt_tst,
 # for example: common/aggs/tst.neglquant.d, and specify tests to be
 # skipped.
 sub load_exceptions {
 	my($listfile) = @_;
 	my($line) = "";
 
 	%exceptions = ();
 	if (length($listfile) > 0) {
 		exit(123) unless open(STDIN, "<$listfile");
 		while (<STDIN>) {
 			chomp;
 			$line = $_;
 			# line is non-empty and not a comment
 			if ((length($line) > 0) && ($line =~ /^\s*[^\s#]/ )) {
 				$exceptions{trim($line)} = 1;
 			}
 		}
 	}
 }
 
 # Return 1 if the test is found in the exception set, 0 otherwise.
 sub is_exception {
 	my($file) = @_;
 	my($i) = -1;
 
 	if (scalar(keys(%exceptions)) == 0) {
 		return 0;
 	}
 
 	# hash absolute pathname after $dt_tst/
 	$file = abs_path($file);
 	$i = index($file, $dt_tst);
 	if ($i == 0) {
 		$file = substr($file, length($dt_tst) + 1);
 		return $exceptions{$file};
 	}
 	return 0;
 }
 
 #
 # Iterate over the set of test files specified on the command-line or by a find
 # on "$defdir/common", "$defdir/$MACH" and "$defdir/$PLATFORM" and execute each
 # one.  If the test file is executable, we fork and exec it. If the test is a
 # .ksh file, we run it with $ksh_path. Otherwise we run dtrace -s on it.  If
 # the file is named tst.* we assume it should return exit status 0.  If the
 # file is named err.* we assume it should return exit status 1.  If the file is
 # named err.D_[A-Z0-9]+[.*].d we use dtrace -xerrtags and examine stderr to
 # ensure that a matching error tag was produced.  If the file is named
 # drp.[A-Z0-9]+[.*].d we use dtrace -xdroptags and examine stderr to ensure
 # that a matching drop tag was produced.  If any *.out or *.err files are found
 # we perform output comparisons.
 #
 # run_tests takes two arguments: The first is the pathname of the dtrace
 # command to invoke when running the tests. The second is the pathname
 # of a file (may be the empty string) listing tests that ought to be
 # skipped (skipped tests are listed as paths relative to $dt_tst, for
 # example: common/aggs/tst.neglquant.d).
 #
 sub run_tests {
 	my($dtrace, $exceptions_path) = @_;
 	my($passed) = 0;
 	my($bypassed) = 0;
 	my($failed) = $errs;
 	my($total) = 0;
 
 	die "$PNAME: $dtrace not found\n" unless (-x "$dtrace");
 	logmsg($dtrace . "\n");
 
 	load_exceptions($exceptions_path);
 
 	foreach $file (sort @files) {
 		$file =~ m:.*/((.*)\.(\w+)):;
 		$name = $1;
 		$base = $2;
 		$ext = $3;
 		
 		$dir = dirname($file);
 		$isksh = 0;
 		$tag = 0;
 		$droptag = 0;
 
 		if ($name =~ /^tst\./) {
 			$isksh = ($ext eq 'ksh');
 			$status = 0;
 		} elsif ($name =~ /^err\.(D_[A-Z0-9_]+)\./) {
 			$status = 1;
 			$tag = $1;
 		} elsif ($name =~ /^err\./) {
 			$status = 1;
 		} elsif ($name =~ /^drp\.([A-Z0-9_]+)\./) {
 			$status = 0;
 			$droptag = $1;
 		} else {
 			errmsg("ERROR: $file is not a valid test file name\n");
 			next;
 		}
 
 		$fullname = "$dir/$name";
-		$exe = "./$base.exe";
+		$exe = "$dir/$base.exe";
 		$exe_pid = -1;
 
 		if ($opt_a && ($status != 0 || $tag != 0 || $droptag != 0 ||
 		    -x $exe || $isksh || -x $fullname)) {
 			$bypassed++;
 			next;
 		}
 
 		if (!$opt_f && is_exception("$dir/$name")) {
 			$bypassed++;
 			next;
 		}
 
 		if (!$isksh && -x $exe) {
 			if (($exe_pid = fork()) == -1) {
 				errmsg(
 				    "ERROR: failed to fork to run $exe: $!\n");
 				next;
 			}
 
 			if ($exe_pid == 0) {
 				open(STDIN, '</dev/null');
 
 				exec($exe);
 
 				warn "ERROR: failed to exec $exe: $!\n";
 			}
 		}
 
 		logmsg("testing $file ... ");
 
 		if (($pid = fork()) == -1) {
 			errmsg("ERROR: failed to fork to run test $file: $!\n");
 			next;
 		}
 
 		if ($pid == 0) {
 			open(STDIN, '</dev/null');
 			exit(125) unless open(STDOUT, ">$opt_d/$$.out");
 			exit(125) unless open(STDERR, ">$opt_d/$$.err");
 
 			unless (chdir($dir)) {
 				warn "ERROR: failed to chdir for $file: $!\n";
 				exit(126);
 			}
 
 			push(@dtrace_argv, '-xerrtags') if ($tag);
 			push(@dtrace_argv, '-xdroptags') if ($droptag);
 			push(@dtrace_argv, $exe_pid) if ($exe_pid != -1);
 
 			if ($isksh) {
 				exit(123) unless open(STDIN, "<$name");
 				exec("$ksh_path /dev/stdin $dtrace");
 			} elsif (-x $name) {
 				warn "ERROR: $name is executable\n";
 				exit(1);
 			} else {
 				if ($tag == 0 && $status == $0 && $opt_a) {
 					push(@dtrace_argv, '-A');
 				}
 
 				push(@dtrace_argv, '-C');
 				push(@dtrace_argv, '-s');
 				push(@dtrace_argv, $name);
 				exec($dtrace, @dtrace_argv);
 			}
 
 			warn "ERROR: failed to exec for $file: $!\n";
 			exit(127);
 		}
 
 		if (waitpid($pid, 0) == -1) {
 			errmsg("ERROR: timed out waiting for $file\n");
 			kill(9, $exe_pid) if ($exe_pid != -1);
 			kill(9, $pid);
 			next;
 		}
 
 		kill(9, $exe_pid) if ($exe_pid != -1);
 
 		if ($tag == 0 && $status == $0 && $opt_a) {
 			#
 			# We can chuck the earler output.
 			#
 			unlink($pid . '.out');
 			unlink($pid . '.err');
 
 			#
 			# This is an anonymous enabling.  We need to get
 			# the module unloaded.
 			#
 			system("dtrace -ae 1> /dev/null 2> /dev/null");
 			system("svcadm disable -s " .
 			    "svc:/network/nfs/mapid:default");
 			system("modunload -i 0 ; modunload -i 0 ; " .
 			    "modunload -i 0");
 			if (!system("modinfo | grep dtrace")) {
 				warn "ERROR: couldn't unload dtrace\n";
 				system("svcadm enable " . 
 				    "-s svc:/network/nfs/mapid:default");
 				exit(124);
 			}
 
 			#
 			# DTrace is gone.  Now update_drv(1M), and rip
 			# everything out again.
 			#
 			system("update_drv dtrace");
 			system("dtrace -ae 1> /dev/null 2> /dev/null");
 			system("modunload -i 0 ; modunload -i 0 ; " .
 			    "modunload -i 0");
 			if (!system("modinfo | grep dtrace")) {
 				warn "ERROR: couldn't unload dtrace\n";
 				system("svcadm enable " . 
 				    "-s svc:/network/nfs/mapid:default");
 				exit(124);
 			}
 
 			#
 			# Now bring DTrace back in.
 			#
 			system("sync ; sync");
 			system("dtrace -l -n bogusprobe 1> /dev/null " .
 			    "2> /dev/null");
 			system("svcadm enable -s " .
 			    "svc:/network/nfs/mapid:default");
 
 			#
 			# That should have caused DTrace to reload with
 			# the new configuration file.  Now we can try to
 			# snag our anonymous state.
 			#
 			if (($pid = fork()) == -1) {
 				errmsg("ERROR: failed to fork to run " .
 				    "test $file: $!\n");
 				next;
 			}
 
 			if ($pid == 0) {
 				open(STDIN, '</dev/null');
 				exit(125) unless open(STDOUT, ">$opt_d/$$.out");
 				exit(125) unless open(STDERR, ">$opt_d/$$.err");
 
 				push(@dtrace_argv, '-a');
 
 				unless (chdir($dir)) {
 					warn "ERROR: failed to chdir " .
 					    "for $file: $!\n";
 					exit(126);
 				}
 
 				exec($dtrace, @dtrace_argv);
 				warn "ERROR: failed to exec for $file: $!\n";
 				exit(127);
 			}
 
 			if (waitpid($pid, 0) == -1) {
 				errmsg("ERROR: timed out waiting for $file\n");
 				kill(9, $pid);
 				next;
 			}
 		}
 
 		logmsg("[$pid]\n");
 		$wstat = $?;
 		$wifexited = ($wstat & 0xFF) == 0;
 		$wexitstat = ($wstat >> 8) & 0xFF;
 		$wtermsig = ($wstat & 0x7F);
 
 		if (!$wifexited) {
 			fail("died from signal $wtermsig");
 			next;
 		}
 
 		if ($wexitstat == 125) {
 			die "$PNAME: failed to create output file in $opt_d " .
 			    "(cd elsewhere or use -d)\n";
 		}
 
 		if ($wexitstat != $status) {
 			fail("returned $wexitstat instead of $status");
 			next;
 		}
 
 		if (-f "$file.out" &&
 		    system("cmp -s $file.out $opt_d/$pid.out") != 0) {
 			fail("stdout mismatch", "$pid.out");
 			next;
 		}
 
 		if (-f "$file.err" &&
 		    system("cmp -s $file.err $opt_d/$pid.err") != 0) {
 			fail("stderr mismatch: see $pid.err");
 			next;
 		}
 
 		if ($tag) {
 			open(TSTERR, "<$opt_d/$pid.err");
 			$tsterr = <TSTERR>;
 			close(TSTERR);
 
 			unless ($tsterr =~ /: \[$tag\] line \d+:/) {
 				fail("errtag mismatch: see $pid.err");
 				next;
 			}
 		}
 
 		if ($droptag) {
 			$found = 0;
 			open(TSTERR, "<$opt_d/$pid.err");
 
 			while (<TSTERR>) {
 				if (/\[$droptag\] /) {
 					$found = 1;
 					last;
 				}
 			}
 
 			close (TSTERR);
 
 			unless ($found) {
 				fail("droptag mismatch: see $pid.err");
 				next;
 			}
 		}
 
 		unless ($opt_s) {
 			unlink($pid . '.out');
 			unlink($pid . '.err');
 		}
 	}
 
 	if ($opt_a) {
 		#
 		# If we're running with anonymous enablings, we need to
 		# restore the .conf file.
 		#
 		system("dtrace -A 1> /dev/null 2> /dev/null");
 		system("dtrace -ae 1> /dev/null 2> /dev/null");
 		system("modunload -i 0 ; modunload -i 0 ; modunload -i 0");
 		system("update_drv dtrace");
 	}
 
 	$total = scalar(@files);
 	$failed = $errs - $failed;
 	$passed = ($total - $failed - $bypassed);
 	$results{$dtrace} = {
 		"passed" => $passed,
 		"bypassed" => $bypassed,
 		"failed" => $failed,
 		"total" => $total
 	};
 }
 
 die $USAGE unless (getopts($OPTSTR));
 usage() if ($opt_h);
 
 foreach $arg (@ARGV) {
 	if (-f $arg) {
 		push(@files, $arg);
 	} elsif (-d $arg) {
 		find(\&wanted, $arg);
 	} else {
 		die "$PNAME: $arg is not a valid file or directory\n";
 	}
 }
 
 $dt_tst = '/opt/SUNWdtrt/tst';
 $dt_bin = '/opt/SUNWdtrt/bin';
 $defdir = -d $dt_tst ? $dt_tst : '.';
 $bindir = -d $dt_bin ? $dt_bin : '.';
 
 find(\&wanted, "$defdir/common") if (scalar(@ARGV) == 0);
 find(\&wanted, "$defdir/$MACH") if (scalar(@ARGV) == 0);
 find(\&wanted, "$defdir/$PLATFORM") if (scalar(@ARGV) == 0);
 die $USAGE if (scalar(@files) == 0);
 
 $dtrace_path = '/usr/sbin/dtrace';
 $jdtrace_path = "$bindir/jdtrace";
 
 %exception_lists = ("$jdtrace_path" => "$bindir/exception.lst");
 
 if ($opt_j || $opt_n || $opt_i) {
 	@dtrace_cmds = ();
 	push(@dtrace_cmds, $dtrace_path) if ($opt_n);
 	push(@dtrace_cmds, $jdtrace_path) if ($opt_j);
 	push(@dtrace_cmds, "/usr/sbin/$opt_i/dtrace") if ($opt_i);
 } else {
 	@dtrace_cmds = ($dtrace_path, $jdtrace_path);
 }
 
 if ($opt_d) {
 	die "$PNAME: -d arg must be absolute path\n" unless ($opt_d =~ /^\//);
 	die "$PNAME: -d arg $opt_d is not a directory\n" unless (-d "$opt_d");
 	system("coreadm -p $opt_d/%p.core");
 } else {
 	my $dir = getcwd;
 	system("coreadm -p $dir/%p.core");
 	$opt_d = '.';
 }
 
 if ($opt_x) {
 	push(@dtrace_argv, '-x');
 	push(@dtrace_argv, $opt_x);
 }
 
 die "$PNAME: failed to open $PNAME.$$.log: $!\n"
     unless (!$opt_l || open(LOG, ">$PNAME.$$.log"));
 
 $ENV{'DTRACE_DEBUG_REGSET'} = 'true';
 
 if ($opt_g) {
 	$ENV{'UMEM_DEBUG'} = 'default,verbose';
 	$ENV{'UMEM_LOGGING'} = 'fail,contents';
 	$ENV{'LD_PRELOAD'} = 'libumem.so';
 }
 
 #
 # Ensure that $PATH contains a cc(1) so that we can execute the
 # test programs that require compilation of C code.
 #
 #$ENV{'PATH'} = $ENV{'PATH'} . ':/ws/onnv-tools/SUNWspro/SS11/bin';
 
 if ($opt_b) {
 	logmsg("badioctl'ing ... ");
 
 	if (($badioctl = fork()) == -1) {
 		errmsg("ERROR: failed to fork to run badioctl: $!\n");
 		next;
 	}
 
 	if ($badioctl == 0) {
 		open(STDIN, '</dev/null');
 		exit(125) unless open(STDOUT, ">$opt_d/$$.out");
 		exit(125) unless open(STDERR, ">$opt_d/$$.err");
 
 		exec($bindir . "/badioctl");
 		warn "ERROR: failed to exec badioctl: $!\n";
 		exit(127);
 	}
 
 
 	logmsg("[$badioctl]\n");
 
 	#
 	# If we're going to be bad, we're just going to iterate over each
 	# test file.
 	#
 	foreach $file (sort @files) {
 		($name = $file) =~ s:.*/::;
 		$dir = dirname($file);
 
 		if (!($name =~ /^tst\./ && $name =~ /\.d$/)) {
 			next;
 		}
 
 		logmsg("baddof'ing $file ... ");
 
 		if (($pid = fork()) == -1) {
 			errmsg("ERROR: failed to fork to run baddof: $!\n");
 			next;
 		}
 
 		if ($pid == 0) {
 			open(STDIN, '</dev/null');
 			exit(125) unless open(STDOUT, ">$opt_d/$$.out");
 			exit(125) unless open(STDERR, ">$opt_d/$$.err");
 
 			unless (chdir($dir)) {
 				warn "ERROR: failed to chdir for $file: $!\n";
 				exit(126);
 			}
 
 			exec($bindir . "/baddof", $name);
 
 			warn "ERROR: failed to exec for $file: $!\n";
 			exit(127);
 		}
 
 		sleep 60;
 		kill(9, $pid);
 		waitpid($pid, 0);
 
 		logmsg("[$pid]\n");
 
 		unless ($opt_s) {
 			unlink($pid . '.out');
 			unlink($pid . '.err');
 		}
 	}
 
 	kill(9, $badioctl);
 	waitpid($badioctl, 0);
 
 	unless ($opt_s) {
 		unlink($badioctl . '.out');
 		unlink($badioctl . '.err');
 	}
 
 	exit(0);
 }
 
 #
 # Run all the tests specified on the command-line (the entire test suite
 # by default) once for each dtrace command tested, skipping any tests
 # not valid for that command. 
 #
 foreach $dtrace_cmd (@dtrace_cmds) {
 	run_tests($dtrace_cmd, $exception_lists{$dtrace_cmd});
 }
 
 $opt_q = 0; # force final summary to appear regardless of -q option
 
 logmsg("\n==== TEST RESULTS ====\n");
 foreach $key (keys %results) {
 	my $passed = $results{$key}{"passed"};
 	my $bypassed = $results{$key}{"bypassed"};
 	my $failed = $results{$key}{"failed"};
 	my $total = $results{$key}{"total"};
 
 	logmsg("\n     mode: " . $key . "\n");
 	logmsg("   passed: " . $passed . "\n");
 	if ($bypassed) {
 		logmsg(" bypassed: " . $bypassed . "\n");
 	}
 	logmsg("   failed: " . $failed . "\n");
 	logmsg("    total: " . $total . "\n");
 }
 
 exit($errs != 0);
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/inline/err.D_OP_INCOMPAT.baddef1.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/inline/err.D_OP_INCOMPAT.baddef1.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/inline/err.D_OP_INCOMPAT.baddef1.d	(revision 277945)
@@ -1,41 +1,41 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *
  * Attempt to create an invalid inline definition by creating an
  * inline of a function type.  This should fail to compile.
  *
  * SECTION: Type and Constant Definitions/Inlines
  *
  * NOTES:
  *
  */
 
-inline cyc_func_t i = "i am a cyclic function";
+inline dtrace_trap_func_t i = "i am a dtrace trap function";
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/inline/err.D_OP_INCOMPAT.badxlate.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/inline/err.D_OP_INCOMPAT.badxlate.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/inline/err.D_OP_INCOMPAT.badxlate.d	(revision 277945)
@@ -1,41 +1,41 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *
  * Test an invalid inline definition of a translator.  An inlined translation
  * must have the same type as the translator output.
  *
  * SECTION: Type and Constant Definitions/Inlines
  *
  * NOTES:
  *
  */
 
-inline vfs_t *invalid = xlate<psinfo_t>(curthread->t_procp);
+inline struct vnode *invalid = xlate<psinfo_t>(curthread->td_proc);
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/json/tst.usdt.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/json/tst.usdt.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/json/tst.usdt.c	(revision 277945)
@@ -1,61 +1,63 @@
 /*
  * This file and its contents are supplied under the terms of the
  * Common Development and Distribution License ("CDDL"), version 1.0.
  * You may only use this file in accordance with the terms of version
  * 1.0 of the CDDL.
  *
  * A full copy of the text of the CDDL should have accompanied this
  * source.  A copy of the CDDL is also available via the Internet at
  * http://www.illumos.org/license/CDDL.
  */
 
 /*
  * Copyright 2012 (c), Joyent, Inc.  All rights reserved.
  */
 
 #include <sys/sdt.h>
+#include <stdio.h>
+#include <stdlib.h>
 #include "usdt.h"
 
 #define	FMT	"{" \
 		"  \"sizes\": [ \"first\", 2, %f ]," \
 		"  \"index\": %d," \
 		"  \"facts\": {" \
 		"    \"odd\": \"%s\"," \
 		"    \"even\": \"%s\"" \
 		"  }," \
 		"  \"action\": \"%s\"" \
 		"}\n"
 
 int
 waiting(volatile int *a)
 {
 	return (*a);
 }
 
 int
 main(int argc, char **argv)
 {
 	volatile int a = 0;
 	int idx;
 	double size = 250.5;
 
 	while (waiting(&a) == 0)
 		continue;
 
 	for (idx = 0; idx < 10; idx++) {
 		char *odd, *even, *json, *action;
 
 		size *= 1.78;
 		odd = idx % 2 == 1 ? "true" : "false";
 		even = idx % 2 == 0 ? "true" : "false";
 		action = idx == 7 ? "ignore" : "print";
 
 		asprintf(&json, FMT, size, idx, odd, even, action);
 		BUNYAN_FAKE_LOG_DEBUG(json);
 		free(json);
 	}
 
 	BUNYAN_FAKE_LOG_DEBUG("{\"finished\": true}");
 
 	return (0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_CHR_NL.char.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_CHR_NL.char.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_CHR_NL.char.d	(revision 277945)
@@ -1,43 +1,45 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma	ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *	Newlines may not be embedded in character constants..
  *
  * SECTION: Lexer
  */
 
 
 BEGIN
 {
-
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Winvalid-pp-token"
 	h = '
 		';
+#pragma clang diagnostic pop
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_CHR_NULL.char.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_CHR_NULL.char.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_CHR_NULL.char.d	(revision 277945)
@@ -1,42 +1,44 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma	ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *	Char constants may not be empty
  *
  * SECTION: Lexer
  */
 
 
 BEGIN
 {
-
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Winvalid-pp-token"
 	h = '';
 	exit(0);
+#pragma clang diagnostic pop
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_STR_NL.string.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_STR_NL.string.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/lexer/err.D_STR_NL.string.d	(revision 277945)
@@ -1,44 +1,46 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma	ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *	Newlines may not be embedded in string literals.
  *
  * SECTION: Lexer
  */
 
 
 BEGIN
 {
-
+#pragma clang diagnostic push
+#pragma clang diagnostic ignored "-Winvalid-pp-token"
 	h = "hello
 
 		there";
 	exit(0);
+#pragma clang diagnostic pop
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/misc/tst.roch.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/misc/tst.roch.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/misc/tst.roch.d	(revision 277945)
@@ -1,93 +1,93 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION: test for assertion failure in the ring buffer code
  *
  * SECTION: Buffers and Buffering/ring Policy; Misc
  */
 
 #pragma ident	"@(#)tst.roch.d	1.2	03/08/11 SMI"
 
 /*
  * A script from Roch Bourbonnais that induced an assertion failure in the
  * ring buffer code.
  */
 #pragma D option strsize=16
 #pragma D option bufsize=10K
 #pragma D option bufpolicy=ring
 
 fbt:::entry
 /(self->done == 0) && (curthread->t_cpu->cpu_intr_actv == 0) /
 {
 	self->done = 1;
 	printf(" %u 0x%llX %d %d comm:%s csathr:%lld", timestamp,
 	    (long long)curthread, pid, tid,
 	    execname, (long long)stackdepth);
 	stack(20);
 }
 
 fbt:::return
 /(self->done == 0) && (curthread->t_cpu->cpu_intr_actv == 0) /
 {
 	self->done = 1;
 	printf(" %u 0x%llX %d %d comm:%s csathr:%lld", timestamp,
 	    (long long) curthread, pid, tid,
 	    execname, (long long) stackdepth);
 	stack(20);
 }
 
 fbt:::entry
 {
 	printf(" %u 0x%llX %d %d ", timestamp,
 	    (long long)curthread, pid, tid);
 }
 
 fbt:::return
 {
 	printf(" %u 0x%llX %d %d tag:%d off:%d ", timestamp,
 	    (long long)curthread, pid, tid, (int)arg1, (int)arg0);
 }
 
-mutex_enter:adaptive-acquire
+mtx_lock:adaptive-acquire
 {
 	printf(" %u 0x%llX %d %d lock:0x%llX", timestamp,
 	    (long long)curthread, pid, tid, arg0);
 }
 
-mutex_exit:adaptive-release
+mtx_unlock:adaptive-release
 {
 	printf(" %u 0x%llX %d %d lock:0x%llX", timestamp,
 	    (long long) curthread, pid, tid, arg0);
 }
 
 tick-1sec
 /n++ == 10/
 {
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/nfs/tst.call3.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/nfs/tst.call3.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/nfs/tst.call3.c	(revision 277945)
@@ -1,441 +1,442 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2008 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <strings.h>
 #include <rpc/rpc.h>
+#include <stdio.h>
 #include <stdlib.h>
 #include <sys/param.h>
 #include <rpcsvc/mount.h>
 
 #include "rpcsvc/nfs_prot.h"
 
 char sharedpath[MAXPATHLEN];
 fhandle3 *rootfh;
 
 /*
  * The waiting() function returns the value passed in, until something
  * external modifies it.  In this case, the D script tst.call.d will
  * modify the value of *a, and thus break the while loop in dotest().
  *
  * This serves the purpose of not making the RPC calls until tst.call.d
  * is active.  Thus, the probes in tst.call.d can fire as a result of
  * the RPC call in dotest().
  */
 
 int
 waiting(volatile int *a)
 {
 	return (*a);
 }
 
 static void
 getattr_arginit(void *argp)
 {
 	GETATTR3args *args = argp;
 
 	args->object.data.data_len = rootfh->fhandle3_len;
 	args->object.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 setattr_arginit(void *argp)
 {
 	SETATTR3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->object.data.data_len = rootfh->fhandle3_len;
 	args->object.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 lookup_arginit(void *argp)
 {
 	LOOKUP3args *args = argp;
 
 	args->what.name = "giant-skunk";
 	args->what.dir.data.data_len = rootfh->fhandle3_len;
 	args->what.dir.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 access_arginit(void *argp)
 {
 	ACCESS3args *args = argp;
 
 	args->object.data.data_len = rootfh->fhandle3_len;
 	args->object.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 commit_arginit(void *argp)
 {
 	COMMIT3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->file.data.data_len = rootfh->fhandle3_len;
 	args->file.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 create_arginit(void *argp)
 {
 	CREATE3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->where.name = "pinky-blue";
 	args->where.dir.data.data_len = rootfh->fhandle3_len;
 	args->where.dir.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 fsinfo_arginit(void *argp)
 {
 	FSINFO3args *args = argp;
 
 	args->fsroot.data.data_len = rootfh->fhandle3_len;
 	args->fsroot.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 fsstat_arginit(void *argp)
 {
 	FSSTAT3args *args = argp;
 
 	args->fsroot.data.data_len = rootfh->fhandle3_len;
 	args->fsroot.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 link_arginit(void *argp)
 {
 	LINK3args *args = argp;
 
 	args->file.data.data_len = rootfh->fhandle3_len;
 	args->file.data.data_val = rootfh->fhandle3_val;
 	args->link.dir.data.data_len = rootfh->fhandle3_len;
 	args->link.dir.data.data_val = rootfh->fhandle3_val;
 	args->link.name = "samf";
 }
 
 static void
 mkdir_arginit(void *argp)
 {
 	MKDIR3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->where.dir.data.data_len = rootfh->fhandle3_len;
 	args->where.dir.data.data_val = rootfh->fhandle3_val;
 	args->where.name = "cookie";
 }
 
 static void
 mknod_arginit(void *argp)
 {
 	MKNOD3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->where.dir.data.data_len = rootfh->fhandle3_len;
 	args->where.dir.data.data_val = rootfh->fhandle3_val;
 	args->where.name = "pookie";
 }
 
 static void
 null_arginit(void *argp)
 {
 }
 
 static void
 pathconf_arginit(void *argp)
 {
 	PATHCONF3args *args = argp;
 
 	args->object.data.data_len = rootfh->fhandle3_len;
 	args->object.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 read_arginit(void *argp)
 {
 	READ3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->file.data.data_len = rootfh->fhandle3_len;
 	args->file.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 readdir_arginit(void *argp)
 {
 	READDIR3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->dir.data.data_len = rootfh->fhandle3_len;
 	args->dir.data.data_val = rootfh->fhandle3_val;
 	args->count = 1024;
 }
 
 static void
 readdirplus_arginit(void *argp)
 {
 	READDIRPLUS3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->dir.data.data_len = rootfh->fhandle3_len;
 	args->dir.data.data_val = rootfh->fhandle3_val;
 	args->dircount = 1024;
 	args->maxcount = 1024;
 }
 
 static void
 readlink_arginit(void *argp)
 {
 	READLINK3args *args = argp;
 
 	args->symlink.data.data_len = rootfh->fhandle3_len;
 	args->symlink.data.data_val = rootfh->fhandle3_val;
 }
 
 static void
 remove_arginit(void *argp)
 {
 	REMOVE3args *args = argp;
 
 	args->object.dir.data.data_len = rootfh->fhandle3_len;
 	args->object.dir.data.data_val = rootfh->fhandle3_val;
 	args->object.name = "antelope";
 }
 
 static void
 rename_arginit(void *argp)
 {
 	RENAME3args *args = argp;
 
 	args->from.dir.data.data_len = rootfh->fhandle3_len;
 	args->from.dir.data.data_val = rootfh->fhandle3_val;
 	args->from.name = "walter";
 	args->to.dir.data.data_len = rootfh->fhandle3_len;
 	args->to.dir.data.data_val = rootfh->fhandle3_val;
 	args->to.name = "wendy";
 }
 
 static void
 rmdir_arginit(void *argp)
 {
 	RMDIR3args *args = argp;
 
 	args->object.dir.data.data_len = rootfh->fhandle3_len;
 	args->object.dir.data.data_val = rootfh->fhandle3_val;
 	args->object.name = "bunny";
 }
 
 static void
 symlink_arginit(void *argp)
 {
 	SYMLINK3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->where.dir.data.data_len = rootfh->fhandle3_len;
 	args->where.dir.data.data_val = rootfh->fhandle3_val;
 	args->where.name = "parlor";
 	args->symlink.symlink_data = "interior";
 }
 
 static void
 write_arginit(void *argp)
 {
 	WRITE3args *args = argp;
 
 	bzero(args, sizeof (*args));
 	args->file.data.data_len = rootfh->fhandle3_len;
 	args->file.data.data_val = rootfh->fhandle3_val;
 }
 
 typedef void (*call3_arginit_t)(void *);
 
 typedef struct {
 	call3_arginit_t arginit;
 	rpcproc_t proc;
 	xdrproc_t xdrargs;
 	size_t argsize;
 	xdrproc_t xdrres;
 	size_t ressize;
 } call3_test_t;
 call3_test_t call3_tests[] = {
 	{getattr_arginit, NFSPROC3_GETATTR, xdr_GETATTR3args,
 	    sizeof (GETATTR3args), xdr_GETATTR3res, sizeof (GETATTR3res)},
 	{setattr_arginit, NFSPROC3_SETATTR, xdr_SETATTR3args,
 	    sizeof (SETATTR3args), xdr_SETATTR3res, sizeof (SETATTR3res)},
 	{lookup_arginit, NFSPROC3_LOOKUP, xdr_LOOKUP3args,
 	    sizeof (LOOKUP3args), xdr_LOOKUP3res, sizeof (LOOKUP3res)},
 	{access_arginit, NFSPROC3_ACCESS, xdr_ACCESS3args,
 	    sizeof (ACCESS3args), xdr_ACCESS3res, sizeof (ACCESS3res)},
 	{commit_arginit, NFSPROC3_COMMIT, xdr_COMMIT3args,
 	    sizeof (COMMIT3args), xdr_COMMIT3res, sizeof (COMMIT3res)},
 	{create_arginit, NFSPROC3_CREATE, xdr_CREATE3args,
 	    sizeof (CREATE3args), xdr_CREATE3res, sizeof (CREATE3res)},
 	{fsinfo_arginit, NFSPROC3_FSINFO, xdr_FSINFO3args,
 	    sizeof (FSINFO3args), xdr_FSINFO3res, sizeof (FSINFO3res)},
 	{fsstat_arginit, NFSPROC3_FSSTAT, xdr_FSSTAT3args,
 	    sizeof (FSSTAT3args), xdr_FSSTAT3res, sizeof (FSSTAT3res)},
 	{link_arginit, NFSPROC3_LINK, xdr_LINK3args,
 	    sizeof (LINK3args), xdr_LINK3res, sizeof (LINK3res)},
 	{mkdir_arginit, NFSPROC3_MKDIR, xdr_MKDIR3args,
 	    sizeof (MKDIR3args), xdr_MKDIR3res, sizeof (MKDIR3res)},
 	{mknod_arginit, NFSPROC3_MKNOD, xdr_MKNOD3args,
 	    sizeof (MKNOD3args), xdr_MKNOD3res, sizeof (MKNOD3res)},
 	/*
 	 * NULL proc is special.  Rather than special case its zero-sized
 	 * args/results, we give it a small nonzero size, so as to not
 	 * make realloc() do the wrong thing.
 	 */
 	{null_arginit, NFSPROC3_NULL, xdr_void, sizeof (int), xdr_void,
 	    sizeof (int)},
 	{pathconf_arginit, NFSPROC3_PATHCONF, xdr_PATHCONF3args,
 	    sizeof (PATHCONF3args), xdr_PATHCONF3res, sizeof (PATHCONF3res)},
 	{read_arginit, NFSPROC3_READ, xdr_READ3args,
 	    sizeof (READ3args), xdr_READ3res, sizeof (READ3res)},
 	{readdir_arginit, NFSPROC3_READDIR, xdr_READDIR3args,
 	    sizeof (READDIR3args), xdr_READDIR3res, sizeof (READDIR3res)},
 	{readdirplus_arginit, NFSPROC3_READDIRPLUS, xdr_READDIRPLUS3args,
 	    sizeof (READDIRPLUS3args), xdr_READDIRPLUS3res,
 	    sizeof (READDIRPLUS3res)},
 	{readlink_arginit, NFSPROC3_READLINK, xdr_READLINK3args,
 	    sizeof (READLINK3args), xdr_READLINK3res, sizeof (READLINK3res)},
 	{remove_arginit, NFSPROC3_REMOVE, xdr_REMOVE3args,
 	    sizeof (REMOVE3args), xdr_REMOVE3res, sizeof (REMOVE3res)},
 	{rename_arginit, NFSPROC3_RENAME, xdr_RENAME3args,
 	    sizeof (RENAME3args), xdr_RENAME3res, sizeof (RENAME3res)},
 	{rmdir_arginit, NFSPROC3_RMDIR, xdr_RMDIR3args,
 	    sizeof (RMDIR3args), xdr_RMDIR3res, sizeof (RMDIR3res)},
 	{symlink_arginit, NFSPROC3_SYMLINK, xdr_SYMLINK3args,
 	    sizeof (SYMLINK3args), xdr_SYMLINK3res, sizeof (SYMLINK3res)},
 	{write_arginit, NFSPROC3_WRITE, xdr_WRITE3args,
 	    sizeof (WRITE3args), xdr_WRITE3res, sizeof (WRITE3res)},
 	{NULL}
 };
 
 int
 dotest(void)
 {
 	CLIENT *client, *mountclient;
 	AUTH *auth;
 	struct timeval timeout;
 	caddr_t args, res;
 	enum clnt_stat status;
 	rpcproc_t proc;
 	call3_test_t *test;
 	void *argbuf = NULL;
 	void *resbuf = NULL;
 	struct mountres3 mountres3;
 	char *sp;
 	volatile int a = 0;
 
 	while (waiting(&a) == 0)
 		continue;
 
 	timeout.tv_sec = 30;
 	timeout.tv_usec = 0;
 
 	mountclient = clnt_create("localhost", MOUNTPROG, MOUNTVERS3, "tcp");
 	if (mountclient == NULL) {
 		clnt_pcreateerror("clnt_create mount");
 		return (1);
 	}
 	auth = authsys_create_default();
 	mountclient->cl_auth = auth;
 	sp = sharedpath;
 	bzero(&mountres3, sizeof (mountres3));
 	status = clnt_call(mountclient, MOUNTPROC_MNT,
 	    xdr_dirpath, (char *)&sp,
 	    xdr_mountres3, (char *)&mountres3,
 	    timeout);
 	if (status != RPC_SUCCESS) {
 		clnt_perror(mountclient, "mnt");
 		return (1);
 	}
 	if (mountres3.fhs_status != 0) {
 		fprintf(stderr, "MOUNTPROG/MOUNTVERS3 failed %d\n",
 		    mountres3.fhs_status);
 		return (1);
 	}
 	rootfh = &mountres3.mountres3_u.mountinfo.fhandle;
 
 	client = clnt_create("localhost", NFS3_PROGRAM, NFS_V3, "tcp");
 	if (client == NULL) {
 		clnt_pcreateerror("clnt_create");
 		return (1);
 	}
 	client->cl_auth = auth;
 
 	for (test = call3_tests; test->arginit; ++test) {
 		argbuf = realloc(argbuf, test->argsize);
 		resbuf = realloc(resbuf, test->ressize);
 		if ((argbuf == NULL) || (resbuf == NULL)) {
 			perror("realloc() failed");
 			return (1);
 		}
 		(test->arginit)(argbuf);
 		bzero(resbuf, test->ressize);
 		status = clnt_call(client, test->proc,
 		    test->xdrargs, argbuf,
 		    test->xdrres, resbuf,
 		    timeout);
 		if (status != RPC_SUCCESS)
 			clnt_perror(client, "call");
 	}
 
 	status = clnt_call(mountclient, MOUNTPROC_UMNT,
 	    xdr_dirpath, (char *)&sp,
 	    xdr_void, NULL,
 	    timeout);
 	if (status != RPC_SUCCESS)
 		clnt_perror(mountclient, "umnt");
 
 	return (0);
 }
 
 /*ARGSUSED*/
 int
 main(int argc, char **argv)
 {
 	char shareline[BUFSIZ], unshareline[BUFSIZ];
 	int rc;
 
 	(void) snprintf(sharedpath, sizeof (sharedpath),
 	    "/tmp/nfsv3test.%d", getpid());
 	(void) snprintf(shareline, sizeof (shareline),
 	    "mkdir %s ; share %s", sharedpath, sharedpath);
 	(void) snprintf(unshareline, sizeof (unshareline),
 	    "unshare %s ; rmdir %s", sharedpath, sharedpath);
 
 	(void) system(shareline);
 	rc = dotest();
 	(void) system(unshareline);
 
 	return (rc);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/offsetof/err.D_UNKNOWN.badmemb.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/offsetof/err.D_UNKNOWN.badmemb.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/offsetof/err.D_UNKNOWN.badmemb.d	(revision 277945)
@@ -1,44 +1,44 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *
  * Test invocation of offsetof() with an invalid member.
  * This should fail at compile time.
  *
  * SECTION: Structs and Unions/Member Sizes and Offsets
  *
  * NOTES:
  *
  */
 
 BEGIN
 {
-	trace(offsetof(vnode_t, v_no_such_member));
+	trace(offsetof(struct vnode, v_no_such_member));
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.args1.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.args1.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.args1.c	(revision 277945)
@@ -1,52 +1,53 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <signal.h>
+#include <stdlib.h>
 #include <unistd.h>
 
 int
 go(int arg0, int arg1, int arg2, int arg3, int arg4, int arg5, int arg6,
     int arg7, int arg8, int arg9)
 {
 	return (arg1);
 }
 
 static void
 handle(int sig)
 {
 	go(0, 1, 2, 3, 4, 5, 6, 7, 8, 9);
 	exit(0);
 }
 
 int
 main(int argc, char **argv)
 {
 	(void) signal(SIGUSR1, handle);
 	for (;;)
 		getpid();
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.fork.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.fork.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.fork.d	(revision 277945)
@@ -1,86 +1,86 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION: make sure fork(2) is okay
  *
  * SECTION: pid provider
  */
 
 #pragma D option destructive
 
 pid$1:a.out:waiting:entry
 {
 	this->value = (int *)alloca(sizeof (int));
 	*this->value = 1;
 	copyout(this->value, arg0, sizeof (int));
 }
 
 proc:::create
 /pid == $1/
 {
-	child = args[0]->pr_pid;
+	child = args[0]->p_pid;
 	trace(pid);
 }
 
 pid$1:a.out:go:
 /pid == child/
 {
 	trace("wrong pid");
 	exit(1);
 }
 
 proc:::exit
 /pid == $1 || pid == child/
 {
 	out++;
 	trace(pid);
 }
 
 proc:::exit
 /out == 2/
 {
 	exit(0);
 }
 
 
 BEGIN
 {
 	/*
 	 * Let's just do this for 5 seconds.
 	 */
 	timeout = timestamp + 5000000000;
 }
 
 profile:::tick-4
 /timestamp > timeout/
 {
 	trace("test timed out");
 	exit(1);
 }
 
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.gcc.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.gcc.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.gcc.c	(revision 277945)
@@ -1,64 +1,66 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
+#include <sys/types.h>
+#include <sys/wait.h>
 #include <spawn.h>
 #include <signal.h>
 #include <stdio.h>
 
 void
 go(void)
 {
 	pid_t pid;
 
 	(void) posix_spawn(&pid, "/bin/ls", NULL, NULL, NULL, NULL);
 
 	(void) waitpid(pid, NULL, 0);
 }
 
 void
 intr(int sig)
 {
 }
 
 int
 main(int argc, char **argv)
 {
 	struct sigaction sa;
 
 	sa.sa_handler = intr;
 	sigfillset(&sa.sa_mask);
 	sa.sa_flags = 0;
 
 	(void) sigaction(SIGUSR1, &sa, NULL);
 
 	for (;;) {
 		go();
 	}
 
 	return (0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.ret1.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.ret1.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.ret1.c	(revision 277945)
@@ -1,64 +1,65 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <signal.h>
+#include <stdlib.h>
 #include <unistd.h>
 
 /*
  * The canonical name should be 'go' since we prefer symbol names with fewer
  * leading underscores.
  */
 
 int a = 100;
 
 int
 help(void)
 {
 	return (a);
 }
 
 int
 go(void)
 {
 	return (help() + 1);
 }
 
 static void
 handle(int sig)
 {
 	go();
 	exit(0);
 }
 
 int
 main(int argc, char **argv)
 {
 	(void) signal(SIGUSR1, handle);
 	for (;;)
 		getpid();
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.ret2.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.ret2.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.ret2.c	(revision 277945)
@@ -1,58 +1,59 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <signal.h>
+#include <stdlib.h>
 #include <unistd.h>
 
 /*
  * The canonical name should be 'go' since we prefer symbol names with fewer
  * leading underscores.
  */
 
 int a = 100;
 
 int
 go(void)
 {
 	return (a);
 }
 
 static void
 handle(int sig)
 {
 	go();
 	exit(0);
 }
 
 int
 main(int argc, char **argv)
 {
 	(void) signal(SIGUSR1, handle);
 	for (;;)
 		getpid();
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.vfork.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.vfork.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.vfork.d	(revision 277945)
@@ -1,61 +1,61 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION: make sure probes called from a vfork(2) child fire in the parent
  *
  * SECTION: pid provider
  */
 
 #pragma D option destructive
 
 pid$1:a.out:waiting:entry
 {
 	this->value = (int *)alloca(sizeof (int));
 	*this->value = 1;
 	copyout(this->value, arg0, sizeof (int));
 }
 
 proc:::create
 /pid == $1/
 {
-	child = args[0]->pr_pid;
+	child = args[0]->p_pid;
 }
 
 pid$1:a.out:go:
 /child != pid/
 {
 	printf("wrong pid (%d %d)", pid, child);
 	exit(1);
 }
 
 syscall::rexit:entry
 /pid == $1/
 {
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.weak1.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.weak1.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.weak1.c	(revision 277945)
@@ -1,58 +1,59 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <signal.h>
+#include <stdlib.h>
 #include <unistd.h>
 
 /*
  * The canonical name should be 'go' since we prefer symbol names with fewer
  * leading underscores.
  */
 
 #pragma weak _go = go
 
 int
 go(int a)
 {
 	return (a + 1);
 }
 
 static void
 handle(int sig)
 {
 	_go(1);
 	exit(0);
 }
 
 int
 main(int argc, char **argv)
 {
 	(void) signal(SIGUSR1, handle);
 	for (;;)
 		getpid();
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.weak2.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.weak2.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/pid/tst.weak2.c	(revision 277945)
@@ -1,58 +1,59 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <signal.h>
+#include <stdlib.h>
 #include <unistd.h>
 
 /*
  * The canonical name should be 'go' since we prefer symbol names with fewer
  * leading underscores.
  */
 
 static int
 go(int a)
 {
 	return (a + 1);
 }
 
 #pragma weak _go = go
 
 static void
 handle(int sig)
 {
 	_go(1);
 	exit(0);
 }
 
 int
 main(int argc, char **argv)
 {
 	(void) signal(SIGUSR1, handle);
 	for (;;)
 		getpid();
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/err.D_PRINT_VOID.bad.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/err.D_PRINT_VOID.bad.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/err.D_PRINT_VOID.bad.d	(revision 277945)
@@ -1,34 +1,34 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright (c) 2012 by Delphix. All rights reserved.
  */
 
 BEGIN
 {
-	print((void)`p0);
+	print((void)`proc0);
 }
 
 BEGIN
 {
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/tst.xlate.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/tst.xlate.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/tst.xlate.d	(revision 277945)
@@ -1,42 +1,42 @@
 /*
  * CDDL HEADER START
  *
  * This file and its contents are supplied under the terms of the
  * Common Development and Distribution License ("CDDL"), version 1.0.
  * You may only use this file in accordance with the terms of version
  * 1.0 of the CDDL.
  *
  * A full copy of the text of the CDDL should have accompanied this
  * source.  A copy of the CDDL is also available via the Internet at
  * http://www.illumos.org/license/CDDL.
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright (c) 2012 by Delphix. All rights reserved.
  */
 
 #pragma D option quiet
 
 typedef struct pancakes {
 	int i;
 	string s;
-	timespec_t t;
+	struct timespec t;
 } pancakes_t;
 
 translator pancakes_t < void *V > {
 	i = 2 * 10;
 	s = strjoin("I like ", "pancakes");
-	t = *(timespec_t *)`dtrace_zero;
+	t = *(struct timespec *)`dtrace_zero;
 };
 
 BEGIN
 {
 	print(*(xlate < pancakes_t * > ((void *)NULL)));
 }
 
 BEGIN
 {
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/tst.xlate.d.out
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/tst.xlate.d.out	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print/tst.xlate.d.out	(revision 277945)
@@ -1,8 +1,8 @@
 pancakes_t {
     int i = 0x14
     string s = [ "I like pancakes" ]
-    timespec_t t = {
+    struct timespec t = {
         time_t tv_sec = 0
         long tv_nsec = 0
     }
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print	(revision 277945)

Property changes on: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/print:r277327-277944
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/probes/tst.probestar.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/probes/tst.probestar.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/probes/tst.probestar.d	(revision 277945)
@@ -1,50 +1,50 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma	ident	"%Z%%M%	%I%	%E% SMI"
 
 /*
  * ASSERTION:
  *
  * Call matching probe clauses with given name using "*"
  *
  * SECTION: Program Structure/Probe Clauses and Declarations
  *
  */
 
 
 #pragma D option quiet
 
 int i;
 BEGIN
 {
 	i = 0;
 }
 
-syscall::*lwp*:entry
+syscall::*wait*:entry
 {
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/proc/tst.discard.ksh
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/proc/tst.discard.ksh	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/proc/tst.discard.ksh	(revision 277945)
@@ -1,74 +1,74 @@
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 # ident	"%Z%%M%	%I%	%E% SMI"
 
 #
 # This script tests that the proc:::signal-discard probe fires correctly
 # and with the correct arguments.
 #
 # If this fails, the script will run indefinitely; it relies on the harness
 # to time it out.
 #
 script()
 {
 	$dtrace -s /dev/stdin <<EOF
 	proc:::signal-discard
-	/args[1]->pr_pid == $child &&
+	/args[1]->p_pid == $child &&
 	    args[1]->pr_psargs == "$longsleep" && args[2] == SIGHUP/
 	{
 		exit(0);
 	}
 EOF
 }
 
 killer()
 {
 	while true; do
 		sleep 1
 		/usr/bin/kill -HUP $child
 	done
 }
 
 if [ $# != 1 ]; then
 	echo expected one argument: '<'dtrace-path'>'
 	exit 2
 fi
 
 dtrace=$1
 longsleep="/usr/bin/sleep 10000"
 
 /usr/bin/nohup $longsleep &
 child=$!
 
 killer &
 killer=$!
 script
 status=$?
 
 kill $child
 kill $killer
 
 exit $status
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/proc/tst.signal.ksh
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/proc/tst.signal.ksh	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/proc/tst.signal.ksh	(revision 277945)
@@ -1,84 +1,84 @@
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 # ident	"%Z%%M%	%I%	%E% SMI"
 
 #
 # This script tests that the proc:::signal-send and proc:::signal-handle
 # probes fire correctly and with the correct arguments.
 #
 # If this fails, the script will run indefinitely; it relies on the harness
 # to time it out.
 #
 script()
 {
 	$dtrace -s /dev/stdin <<EOF
 	proc:::signal-send
 	/execname == "kill" && curpsinfo->pr_ppid == $child &&
 	    args[1]->pr_psargs == "$longsleep" && args[2] == SIGUSR1/
 	{
 		/*
 		 * This is guaranteed to not race with signal-handle.
 		 */
-		target = args[1]->pr_pid;
+		target = args[1]->p_pid;
 	}
 
 	proc:::signal-handle
 	/target == pid && args[0] == SIGUSR1/
 	{
 		exit(0);
 	}
 EOF
 }
 
 sleeper()
 {
 	while true; do
 		$longsleep &
 		sleep 1
 		/usr/bin/kill -USR1 $!
 	done
 }
 
 if [ $# != 1 ]; then
 	echo expected one argument: '<'dtrace-path'>'
 	exit 2
 fi
 
 dtrace=$1
 longsleep="/usr/bin/sleep 10000"
 
 sleeper &
 child=$!
 
 script
 status=$?
 
 pstop $child
 pkill -P $child
 kill $child
 prun $child
 
 exit $status
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.func.ksh
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.func.ksh	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.func.ksh	(revision 277945)
@@ -1,74 +1,74 @@
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 # ident	"%Z%%M%	%I%	%E% SMI"
 
 script()
 {
 	$dtrace -qs /dev/stdin <<EOF
 	profile-1234hz
 	/arg0 != 0/
 	{
 		@[func(arg0)] = count();
 	}
 
 	tick-100ms
 	/i++ == 50/
 	{
 		exit(0);
 	}
 EOF
 }
 
 spinny()
 {
 	while true; do
 		/usr/bin/date > /dev/null
 	done
 }
 
 if [ $# != 1 ]; then
 	echo expected one argument: '<'dtrace-path'>'
 	exit 2
 fi
 
 dtrace=$1
 
 spinny &
 child=$!
 
 #
-# This is gutsy -- we're assuming that mutex_enter(9F) will show up in the
+# This is gutsy -- we're assuming that mtx_lock(9) will show up in the
 # output.  This is most likely _not_ to show up in the output if the 
 # platform does not support arbitrary resolution interval timers -- but
 # the above script was stress-tested down to 100 hertz and still ran
 # successfully on all platforms, so one is hopeful that this test will pass
 # even in that case.
 #
-script | tee /dev/fd/2 | grep mutex_enter > /dev/null
+script | tee /dev/fd/2 | grep mtx_lock > /dev/null
 status=$?
 
 kill $child
 exit $status
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.mod.ksh
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.mod.ksh	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.mod.ksh	(revision 277945)
@@ -1,70 +1,70 @@
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 # ident	"%Z%%M%	%I%	%E% SMI"
 
 script()
 {
 	$dtrace -qs /dev/stdin <<EOF
 	profile-1234hz
 	/arg0 != 0/
 	{
 		@[mod(arg0)] = count();
 	}
 
 	tick-100ms
 	/i++ == 20/
 	{
 		exit(0);
 	}
 EOF
 }
 
 spinny()
 {
 	while true; do
 		/usr/bin/date > /dev/null
 	done
 }
 
 if [ $# != 1 ]; then
 	echo expected one argument: '<'dtrace-path'>'
 	exit 2
 fi
 
 dtrace=$1
 
 spinny &
 child=$!
 
 #
 # The only thing we can be sure of is that some module named "unix" (or
 # "genunix") did some work -- so that's all we'll check.
 #
-script | tee /dev/fd/2 | grep unix > /dev/null
+script | tee /dev/fd/2 | grep kernel > /dev/null
 status=$? 
 
 kill $child
 exit $status
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.sym.ksh
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.sym.ksh	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/profile-n/tst.sym.ksh	(revision 277945)
@@ -1,70 +1,70 @@
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 # ident	"%Z%%M%	%I%	%E% SMI"
 
 script()
 {
 	$dtrace -qs /dev/stdin <<EOF
 	profile-1234hz
 	/arg0 != 0/
 	{
 		@[sym(arg0)] = count();
 	}
 
 	tick-100ms
 	/i++ == 50/
 	{
 		exit(0);
 	}
 EOF
 }
 
 spinny()
 {
 	while true; do
 		/usr/bin/date > /dev/null
 	done
 }
 
 if [ $# != 1 ]; then
 	echo expected one argument: '<'dtrace-path'>'
 	exit 2
 fi
 
 dtrace=$1
 
 spinny &
 child=$!
 
 #
 # This is the same gutsy test as that found in the func() test; see that
 # test for the rationale.
 #
-script | tee /dev/fd/2 | grep mutex_enter > /dev/null
+script | tee /dev/fd/2 | grep mtx_lock > /dev/null
 status=$?
 
 kill $child
 exit $status
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/scalars/tst.selfarray2.d
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/scalars/tst.selfarray2.d	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/scalars/tst.selfarray2.d	(revision 277945)
@@ -1,64 +1,64 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2006 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #pragma D option quiet
 #pragma D option destructive
 #pragma D option dynvarsize=1m
 
 struct bar {
-	int pid;
-	kthread_t *curthread;
+	pid_t pid;
+	struct thread *curthread;
 };
 
 self struct bar foo[int];
 
 syscall:::entry
 /!self->foo[0].pid/
 {
 	self->foo[0].pid = pid;
 	self->foo[0].curthread = curthread;
 }
 
 syscall:::entry
 /self->foo[0].pid != pid/
 {
 	printf("expected %d, found %d (found curthread %p, curthread is %p)\n",
 	    pid, self->foo[0].pid, self->foo[0].curthread, curthread);
 	exit(1);
 }
 
 tick-100hz
 {
 	system("date > /dev/null")
 }
 
 tick-1sec
 /i++ == 10/
 {
 	exit(0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/syscall/tst.args.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/syscall/tst.args.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/syscall/tst.args.c	(revision 277945)
@@ -1,41 +1,42 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
 #include <stdio.h>
 #include <sys/syscall.h>
+#include <unistd.h>
 
 /*ARGSUSED*/
 int
 main(int argc, char **argv)
 {
 	for (;;) {
 		(void) syscall(SYS_mmap, NULL, 1, 2, 3, -1, 0x12345678);
 	}
 
 	return (0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/usdt/tst.dlclose1.ksh
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/usdt/tst.dlclose1.ksh	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/usdt/tst.dlclose1.ksh	(revision 277945)
@@ -1,159 +1,162 @@
 #!/bin/ksh -p
 #
 # CDDL HEADER START
 #
 # The contents of this file are subject to the terms of the
 # Common Development and Distribution License (the "License").
 # You may not use this file except in compliance with the License.
 #
 # You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
 # or http://www.opensolaris.org/os/licensing.
 # See the License for the specific language governing permissions
 # and limitations under the License.
 #
 # When distributing Covered Code, include this CDDL HEADER in each
 # file and include the License file at usr/src/OPENSOLARIS.LICENSE.
 # If applicable, add the following below this CDDL HEADER, with the
 # fields enclosed by brackets "[]" replaced with your own identifying
 # information: Portions Copyright [yyyy] [name of copyright owner]
 #
 # CDDL HEADER END
 #
 
 #
 # Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
 # Use is subject to license terms.
 #
 # ident	"%Z%%M%	%I%	%E% SMI"
 
 #
 # This test verifies that USDT providers are removed when its associated
 # load object is closed via dlclose(3dl).
 #
 
 if [ $# != 1 ]; then
 	echo expected one argument: '<'dtrace-path'>'
 	exit 2
 fi
 
 dtrace=$1
 DIR=/var/tmp/dtest.$$
 
 mkdir $DIR
 cd $DIR
 
 cat > Makefile <<EOF
 all: main livelib.so deadlib.so
 
 main: main.o prov.o
 	cc -o main main.o
 
 main.o: main.c
 	cc -c main.c
 
 
 livelib.so: livelib.o prov.o
 	cc -z defs -G -o livelib.so livelib.o prov.o -lc
 
 livelib.o: livelib.c prov.h
 	cc -c livelib.c
 
 prov.o: livelib.o prov.d
 	$dtrace -G -s prov.d livelib.o
 
 prov.h: prov.d
 	$dtrace -h -s prov.d
 
 
 deadlib.so: deadlib.o
 	cc -z defs -G -o deadlib.so deadlib.o -lc
 
 deadlib.o: deadlib.c
 	cc -c deadlib.c
 
 clean:
 	rm -f main.o livelib.o prov.o prov.h deadlib.o
 
 clobber: clean
 	rm -f main livelib.so deadlib.so
 EOF
 
 cat > prov.d <<EOF
 provider test_prov {
 	probe go();
 };
 EOF
 
 cat > livelib.c <<EOF
 #include "prov.h"
 
 void
 go(void)
 {
 	TEST_PROV_GO();
 }
 EOF
 
 cat > deadlib.c <<EOF
 void
 go(void)
 {
 }
 EOF
 
 
 cat > main.c <<EOF
 #include <dlfcn.h>
 #include <unistd.h>
 #include <stdio.h>
+#include <signal.h>
 
 int
 main(int argc, char **argv)
 {
 	void *live;
+	sigset_t mask;
 
 	if ((live = dlopen("./livelib.so", RTLD_LAZY | RTLD_LOCAL)) == NULL) {
 		printf("dlopen of livelib.so failed: %s\n", dlerror());
 		return (1);
 	}
 
 	(void) dlclose(live);
 
-	pause();
+	(void) sigemptyset(&mask);
+	(void) sigsuspend(&mask);
 
 	return (0);
 }
 EOF
 
 /usr/bin/make > /dev/null
 if [ $? -ne 0 ]; then
 	print -u2 "failed to build"
 	exit 1
 fi
 
 script() {
 	$dtrace -w -x bufsize=1k -c ./main -qs /dev/stdin <<EOF
-	syscall::pause:entry
+	syscall::sigsuspend:entry
 	/pid == \$target/
 	{
 		system("$dtrace -l -P test_prov*");
 		system("kill %d", \$target);
 		exit(0);
 	}
 
 	tick-1s
 	/i++ == 5/
 	{
 		printf("failed\n");
 		exit(1);
 	}
 EOF
 }
 
 script 2>&1
 status=$?
 
 cd /
 /bin/rm -rf $DIR
 
 exit $status
Index: projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/usdt/tst.forker.c
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/usdt/tst.forker.c	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris/cmd/dtrace/test/tst/common/usdt/tst.forker.c	(revision 277945)
@@ -1,47 +1,51 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  */
 
 /*
  * Copyright 2007 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 #pragma ident	"%Z%%M%	%I%	%E% SMI"
 
+#include <sys/types.h>
+#include <sys/wait.h>
+
+#include <stdlib.h>
 #include <unistd.h>
 
 #include "forker.h"
 
 int
 main(int argc, char **argv)
 {
 	int i;
 
 	for (i = 0; i < 10000; i++) {
 		FORKER_FIRE();
 		if (fork() == 0)
 			exit(0);
 
 		(void) wait(NULL);
 	}
 
 	return (0);
 }
Index: projects/clang360-import/cddl/contrib/opensolaris
===================================================================
--- projects/clang360-import/cddl/contrib/opensolaris	(revision 277944)
+++ projects/clang360-import/cddl/contrib/opensolaris	(revision 277945)

Property changes on: projects/clang360-import/cddl/contrib/opensolaris
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/cddl/contrib/opensolaris:r277719-277944
Index: projects/clang360-import/cddl
===================================================================
--- projects/clang360-import/cddl	(revision 277944)
+++ projects/clang360-import/cddl	(revision 277945)

Property changes on: projects/clang360-import/cddl
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/cddl:r277719-277944
Index: projects/clang360-import/contrib/libcxxrt/stdexcept.cc
===================================================================
--- projects/clang360-import/contrib/libcxxrt/stdexcept.cc	(revision 277944)
+++ projects/clang360-import/contrib/libcxxrt/stdexcept.cc	(revision 277945)
@@ -1,104 +1,99 @@
 /* 
  * Copyright 2010-2011 PathScale, Inc. All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions are met:
  *
  * 1. Redistributions of source code must retain the above copyright notice,
  *    this list of conditions and the following disclaimer.
  *
  * 2. Redistributions in binary form must reproduce the above copyright notice,
  *    this list of conditions and the following disclaimer in the documentation
  *    and/or other materials provided with the distribution.
  * 
  * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS ``AS
  * IS'' AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO,
  * THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
  * PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
  * CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
  * EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
  * PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
  * OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY,
  * WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR
  * OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF
  * ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  */
 
 /**
  * stdexcept.cc - provides stub implementations of the exceptions required by the runtime.
  */
 #include "stdexcept.h"
 
 namespace std {
 
 exception::exception() throw() {}
 exception::~exception() {}
 exception::exception(const exception&) throw() {}
 exception& exception::operator=(const exception&) throw()
 {
 	return *this;
 }
 const char* exception::what() const throw()
 {
 	return "std::exception";
 }
 
 bad_alloc::bad_alloc() throw() {}
 bad_alloc::~bad_alloc() {}
 bad_alloc::bad_alloc(const bad_alloc&) throw() {}
 bad_alloc& bad_alloc::operator=(const bad_alloc&) throw()
 {
 	return *this;
 }
 const char* bad_alloc::what() const throw()
 {
 	return "cxxrt::bad_alloc";
 }
 
 
 
 bad_cast::bad_cast() throw() {}
 bad_cast::~bad_cast() {}
 bad_cast::bad_cast(const bad_cast&) throw() {}
 bad_cast& bad_cast::operator=(const bad_cast&) throw()
 {
 	return *this;
 }
 const char* bad_cast::what() const throw()
 {
 	return "std::bad_cast";
 }
 
 bad_typeid::bad_typeid() throw() {}
 bad_typeid::~bad_typeid() {}
 bad_typeid::bad_typeid(const bad_typeid &__rhs) throw() {}
 bad_typeid& bad_typeid::operator=(const bad_typeid &__rhs) throw()
 {
 	return *this;
 }
 
 const char* bad_typeid::what() const throw()
 {
 	return "std::bad_typeid";
 }
 
-__attribute__((weak))
 bad_array_new_length::bad_array_new_length() throw() {}
-__attribute__((weak))
 bad_array_new_length::~bad_array_new_length() {}
-__attribute__((weak))
 bad_array_new_length::bad_array_new_length(const bad_array_new_length&) throw() {}
-__attribute__((weak))
 bad_array_new_length& bad_array_new_length::operator=(const bad_array_new_length&) throw()
 {
 	return *this;
 }
 
-__attribute__((weak))
 const char* bad_array_new_length::what() const throw()
 {
 	return "std::bad_array_new_length";
 }
 
 } // namespace std
 
Index: projects/clang360-import/contrib/libcxxrt
===================================================================
--- projects/clang360-import/contrib/libcxxrt	(revision 277944)
+++ projects/clang360-import/contrib/libcxxrt	(revision 277945)

Property changes on: projects/clang360-import/contrib/libcxxrt
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/contrib/libcxxrt:r277327-277944
Index: projects/clang360-import/lib/libnv/Makefile
===================================================================
--- projects/clang360-import/lib/libnv/Makefile	(revision 277944)
+++ projects/clang360-import/lib/libnv/Makefile	(revision 277945)
@@ -1,168 +1,169 @@
 # $FreeBSD$
 
 SHLIBDIR?= /lib
 
 .include <src.opts.mk>
 
 LIB=	nv
 SHLIB_MAJOR= 0
 
 SRCS=	dnvlist.c
 SRCS+=	msgio.c
 SRCS+=	nvlist.c
 SRCS+=	nvpair.c
 
 INCS=	dnv.h
 INCS+=	nv.h
 
 MAN+=	nv.3
 
 MLINKS+=nv.3 libnv.3 \
 	nv.3 nvlist.3
 MLINKS+=nv.3 nvlist_create.3 \
 	nv.3 nvlist_destroy.3 \
 	nv.3 nvlist_error.3 \
 	nv.3 nvlist_empty.3 \
 	nv.3 nvlist_clone.3 \
 	nv.3 nvlist_dump.3 \
 	nv.3 nvlist_fdump.3 \
 	nv.3 nvlist_size.3 \
 	nv.3 nvlist_pack.3 \
 	nv.3 nvlist_unpack.3 \
 	nv.3 nvlist_send.3 \
 	nv.3 nvlist_recv.3 \
 	nv.3 nvlist_xfer.3 \
 	nv.3 nvlist_next.3 \
 	nv.3 nvlist_exists.3 \
 	nv.3 nvlist_exists_type.3 \
 	nv.3 nvlist_exists_null.3 \
 	nv.3 nvlist_exists_bool.3 \
 	nv.3 nvlist_exists_number.3 \
 	nv.3 nvlist_exists_string.3 \
 	nv.3 nvlist_exists_nvlist.3 \
 	nv.3 nvlist_exists_descriptor.3 \
 	nv.3 nvlist_exists_binary.3 \
 	nv.3 nvlist_add_null.3 \
 	nv.3 nvlist_add_bool.3 \
 	nv.3 nvlist_add_number.3 \
 	nv.3 nvlist_add_string.3 \
 	nv.3 nvlist_add_stringf.3 \
 	nv.3 nvlist_add_stringv.3 \
 	nv.3 nvlist_add_nvlist.3 \
 	nv.3 nvlist_add_descriptor.3 \
 	nv.3 nvlist_add_binary.3 \
 	nv.3 nvlist_move_string.3 \
 	nv.3 nvlist_move_nvlist.3 \
 	nv.3 nvlist_move_descriptor.3 \
 	nv.3 nvlist_move_binary.3 \
 	nv.3 nvlist_get_bool.3 \
 	nv.3 nvlist_get_number.3 \
 	nv.3 nvlist_get_string.3 \
 	nv.3 nvlist_get_nvlist.3 \
 	nv.3 nvlist_get_descriptor.3 \
 	nv.3 nvlist_get_binary.3 \
+	nv.3 nvlist_get_parent.3 \
 	nv.3 nvlist_take_bool.3 \
 	nv.3 nvlist_take_number.3 \
 	nv.3 nvlist_take_string.3 \
 	nv.3 nvlist_take_nvlist.3 \
 	nv.3 nvlist_take_descriptor.3 \
 	nv.3 nvlist_take_binary.3 \
 	nv.3 nvlist_free.3 \
 	nv.3 nvlist_free_type.3 \
 	nv.3 nvlist_free_null.3 \
 	nv.3 nvlist_free_bool.3 \
 	nv.3 nvlist_free_number.3 \
 	nv.3 nvlist_free_string.3 \
 	nv.3 nvlist_free_nvlist.3 \
 	nv.3 nvlist_free_descriptor.3 \
 	nv.3 nvlist_free_binary.3
 MLINKS+=nv.3 nvlist_existsf.3 \
 	nv.3 nvlist_existsf_type.3 \
 	nv.3 nvlist_existsf_null.3 \
 	nv.3 nvlist_existsf_bool.3 \
 	nv.3 nvlist_existsf_number.3 \
 	nv.3 nvlist_existsf_string.3 \
 	nv.3 nvlist_existsf_nvlist.3 \
 	nv.3 nvlist_existsf_descriptor.3 \
 	nv.3 nvlist_existsf_binary.3 \
 	nv.3 nvlist_addf_null.3 \
 	nv.3 nvlist_addf_bool.3 \
 	nv.3 nvlist_addf_number.3 \
 	nv.3 nvlist_addf_string.3 \
 	nv.3 nvlist_addf_nvlist.3 \
 	nv.3 nvlist_addf_descriptor.3 \
 	nv.3 nvlist_addf_binary.3 \
 	nv.3 nvlist_movef_string.3 \
 	nv.3 nvlist_movef_nvlist.3 \
 	nv.3 nvlist_movef_descriptor.3 \
 	nv.3 nvlist_movef_binary.3 \
 	nv.3 nvlist_getf_bool.3 \
 	nv.3 nvlist_getf_number.3 \
 	nv.3 nvlist_getf_string.3 \
 	nv.3 nvlist_getf_nvlist.3 \
 	nv.3 nvlist_getf_descriptor.3 \
 	nv.3 nvlist_getf_binary.3 \
 	nv.3 nvlist_takef_bool.3 \
 	nv.3 nvlist_takef_number.3 \
 	nv.3 nvlist_takef_string.3 \
 	nv.3 nvlist_takef_nvlist.3 \
 	nv.3 nvlist_takef_descriptor.3 \
 	nv.3 nvlist_takef_binary.3 \
 	nv.3 nvlist_freef.3 \
 	nv.3 nvlist_freef_type.3 \
 	nv.3 nvlist_freef_null.3 \
 	nv.3 nvlist_freef_bool.3 \
 	nv.3 nvlist_freef_number.3 \
 	nv.3 nvlist_freef_string.3 \
 	nv.3 nvlist_freef_nvlist.3 \
 	nv.3 nvlist_freef_descriptor.3 \
 	nv.3 nvlist_freef_binary.3
 MLINKS+=nv.3 nvlist_existsv.3 \
 	nv.3 nvlist_existsv_type.3 \
 	nv.3 nvlist_existsv_null.3 \
 	nv.3 nvlist_existsv_bool.3 \
 	nv.3 nvlist_existsv_number.3 \
 	nv.3 nvlist_existsv_string.3 \
 	nv.3 nvlist_existsv_nvlist.3 \
 	nv.3 nvlist_existsv_descriptor.3 \
 	nv.3 nvlist_existsv_binary.3 \
 	nv.3 nvlist_addv_null.3 \
 	nv.3 nvlist_addv_bool.3 \
 	nv.3 nvlist_addv_number.3 \
 	nv.3 nvlist_addv_string.3 \
 	nv.3 nvlist_addv_nvlist.3 \
 	nv.3 nvlist_addv_descriptor.3 \
 	nv.3 nvlist_addv_binary.3 \
 	nv.3 nvlist_movev_string.3 \
 	nv.3 nvlist_movev_nvlist.3 \
 	nv.3 nvlist_movev_descriptor.3 \
 	nv.3 nvlist_movev_binary.3 \
 	nv.3 nvlist_getv_bool.3 \
 	nv.3 nvlist_getv_number.3 \
 	nv.3 nvlist_getv_string.3 \
 	nv.3 nvlist_getv_nvlist.3 \
 	nv.3 nvlist_getv_descriptor.3 \
 	nv.3 nvlist_getv_binary.3 \
 	nv.3 nvlist_takev_bool.3 \
 	nv.3 nvlist_takev_number.3 \
 	nv.3 nvlist_takev_string.3 \
 	nv.3 nvlist_takev_nvlist.3 \
 	nv.3 nvlist_takev_descriptor.3 \
 	nv.3 nvlist_takev_binary.3 \
 	nv.3 nvlist_freev.3 \
 	nv.3 nvlist_freev_type.3 \
 	nv.3 nvlist_freev_null.3 \
 	nv.3 nvlist_freev_bool.3 \
 	nv.3 nvlist_freev_number.3 \
 	nv.3 nvlist_freev_string.3 \
 	nv.3 nvlist_freev_nvlist.3 \
 	nv.3 nvlist_freev_descriptor.3 \
 	nv.3 nvlist_freev_binary.3
 
 WARNS?=	6
 
 .if ${MK_TESTS} != "no"
 SUBDIR+=	tests
 .endif
 
 .include <bsd.lib.mk>
Index: projects/clang360-import/lib/libnv/nv.3
===================================================================
--- projects/clang360-import/lib/libnv/nv.3	(revision 277944)
+++ projects/clang360-import/lib/libnv/nv.3	(revision 277945)
@@ -1,610 +1,632 @@
 .\"
 .\" Copyright (c) 2013 The FreeBSD Foundation
 .\" All rights reserved.
 .\"
 .\" This documentation was written by Pawel Jakub Dawidek under sponsorship
 .\" the FreeBSD Foundation.
 .\"
 .\" Redistribution and use in source and binary forms, with or without
 .\" modification, are permitted provided that the following conditions
 .\" are met:
 .\" 1. Redistributions of source code must retain the above copyright
 .\"    notice, this list of conditions and the following disclaimer.
 .\" 2. Redistributions in binary form must reproduce the above copyright
 .\"    notice, this list of conditions and the following disclaimer in the
 .\"    documentation and/or other materials provided with the distribution.
 .\"
 .\" THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
 .\" ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
 .\" IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
 .\" ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
 .\" FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
 .\" DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
 .\" OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
 .\" HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
 .\" LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
 .\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
 .\" SUCH DAMAGE.
 .\"
 .\" $FreeBSD$
 .\"
-.Dd September 25, 2014
+.Dd January 30, 2015
 .Dt NV 3
 .Os
 .Sh NAME
 .Nm nvlist_create ,
 .Nm nvlist_destroy ,
 .Nm nvlist_error ,
 .Nm nvlist_empty ,
 .Nm nvlist_exists ,
 .Nm nvlist_free ,
 .Nm nvlist_clone ,
 .Nm nvlist_dump ,
 .Nm nvlist_fdump ,
 .Nm nvlist_size ,
 .Nm nvlist_pack ,
 .Nm nvlist_unpack ,
 .Nm nvlist_send ,
 .Nm nvlist_recv ,
 .Nm nvlist_xfer ,
 .Nm nvlist_next ,
 .Nm nvlist_add ,
 .Nm nvlist_move ,
 .Nm nvlist_get ,
 .Nm nvlist_take
 .Nd "library for name/value pairs"
 .Sh LIBRARY
 .Lb libnv
 .Sh SYNOPSIS
 .In nv.h
 .Ft "nvlist_t *"
 .Fn nvlist_create "int flags"
 .Ft void
 .Fn nvlist_destroy "nvlist_t *nvl"
 .Ft int
 .Fn nvlist_error "const nvlist_t *nvl"
 .Ft bool
 .Fn nvlist_empty "const nvlist_t *nvl"
 .\"
 .Ft "nvlist_t *"
 .Fn nvlist_clone "const nvlist_t *nvl"
 .\"
 .Ft void
 .Fn nvlist_dump "const nvlist_t *nvl, int fd"
 .Ft void
 .Fn nvlist_fdump "const nvlist_t *nvl, FILE *fp"
 .\"
 .Ft size_t
 .Fn nvlist_size "const nvlist_t *nvl"
 .Ft "void *"
 .Fn nvlist_pack "const nvlist_t *nvl" "size_t *sizep"
 .Ft "nvlist_t *"
 .Fn nvlist_unpack "const void *buf" "size_t size"
 .\"
 .Ft int
 .Fn nvlist_send "int sock" "const nvlist_t *nvl"
 .Ft "nvlist_t *"
 .Fn nvlist_recv "int sock"
 .Ft "nvlist_t *"
 .Fn nvlist_xfer "int sock" "nvlist_t *nvl"
 .\"
 .Ft "const char *"
 .Fn nvlist_next "const nvlist_t *nvl" "int *typep" "void **cookiep"
 .\"
 .Ft bool
 .Fn nvlist_exists "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_type "const nvlist_t *nvl" "const char *name" "int type"
 .Ft bool
 .Fn nvlist_exists_null "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_bool "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_number "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_string "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_nvlist "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_descriptor "const nvlist_t *nvl" "const char *name"
 .Ft bool
 .Fn nvlist_exists_binary "const nvlist_t *nvl" "const char *name"
 .\"
 .Ft void
 .Fn nvlist_add_null "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_add_bool "nvlist_t *nvl" "const char *name" "bool value"
 .Ft void
 .Fn nvlist_add_number "nvlist_t *nvl" "const char *name" "uint64_t value"
 .Ft void
 .Fn nvlist_add_string "nvlist_t *nvl" "const char *name" "const char *value"
 .Ft void
 .Fn nvlist_add_stringf "nvlist_t *nvl" "const char *name" "const char *valuefmt" "..."
 .Ft void
 .Fn nvlist_add_stringv "nvlist_t *nvl" "const char *name" "const char *valuefmt" "va_list valueap"
 .Ft void
 .Fn nvlist_add_nvlist "nvlist_t *nvl" "const char *name" "const nvlist_t *value"
 .Ft void
 .Fn nvlist_add_descriptor "nvlist_t *nvl" "const char *name" "int value"
 .Ft void
 .Fn nvlist_add_binary "nvlist_t *nvl" "const char *name" "const void *value" "size_t size"
 .\"
 .Ft void
 .Fn nvlist_move_string "nvlist_t *nvl" "const char *name" "char *value"
 .Ft void
 .Fn nvlist_move_nvlist "nvlist_t *nvl" "const char *name" "nvlist_t *value"
 .Ft void
 .Fn nvlist_move_descriptor "nvlist_t *nvl" "const char *name" "int value"
 .Ft void
 .Fn nvlist_move_binary "nvlist_t *nvl" "const char *name" "void *value" "size_t size"
 .\"
 .Ft bool
 .Fn nvlist_get_bool "const nvlist_t *nvl" "const char *name"
 .Ft uint64_t
 .Fn nvlist_get_number "const nvlist_t *nvl" "const char *name"
 .Ft "const char *"
 .Fn nvlist_get_string "const nvlist_t *nvl" "const char *name"
 .Ft "const nvlist_t *"
 .Fn nvlist_get_nvlist "const nvlist_t *nvl" "const char *name"
 .Ft int
 .Fn nvlist_get_descriptor "const nvlist_t *nvl" "const char *name"
 .Ft "const void *"
 .Fn nvlist_get_binary "const nvlist_t *nvl" "const char *name" "size_t *sizep"
 .Ft "const nvlist_t *"
-.Fn nvlist_get_parent "const nvlist_t *nvl"
+.Fn nvlist_get_parent "const nvlist_t *nvl" "void **cookiep"
 .\"
 .Ft bool
 .Fn nvlist_take_bool "nvlist_t *nvl" "const char *name"
 .Ft uint64_t
 .Fn nvlist_take_number "nvlist_t *nvl" "const char *name"
 .Ft "char *"
 .Fn nvlist_take_string "nvlist_t *nvl" "const char *name"
 .Ft "nvlist_t *"
 .Fn nvlist_take_nvlist "nvlist_t *nvl" "const char *name"
 .Ft int
 .Fn nvlist_take_descriptor "nvlist_t *nvl" "const char *name"
 .Ft "void *"
 .Fn nvlist_take_binary "nvlist_t *nvl" "const char *name" "size_t *sizep"
 .\"
 .Ft void
 .Fn nvlist_free "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_type "nvlist_t *nvl" "const char *name" "int type"
 .\"
 .Ft void
 .Fn nvlist_free_null "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_bool "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_number "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_string "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_nvlist "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_descriptor "nvlist_t *nvl" "const char *name"
 .Ft void
 .Fn nvlist_free_binary "nvlist_t *nvl" "const char *name"
 .Sh DESCRIPTION
 The
 .Nm libnv
 library allows to easily manage name value pairs as well as send and receive
 them over sockets.
 A group (list) of name value pairs is called an
 .Nm nvlist .
 The API supports the following data types:
 .Bl -ohang -offset indent
 .It Sy null ( NV_TYPE_NULL )
 There is no data associated with the name.
 .It Sy bool ( NV_TYPE_BOOL )
 The value can be either
 .Dv true
 or
 .Dv false .
 .It Sy number ( NV_TYPE_NUMBER )
 The value is a number stored as
 .Vt uint64_t .
 .It Sy string ( NV_TYPE_STRING )
 The value is a C string.
 .It Sy nvlist ( NV_TYPE_NVLIST )
 The value is a nested nvlist.
 .It Sy descriptor ( NV_TYPE_DESCRIPTOR )
 The value is a file descriptor.
 Note that file descriptors can be sent only over
 .Xr unix 4
 domain sockets.
 .It Sy binary ( NV_TYPE_BINARY )
 The value is a binary buffer.
 .El
 .Pp
 The
 .Fn nvlist_create
 function allocates memory and initializes an nvlist.
 .Pp
 The following flag can be provided:
 .Pp
 .Bl -tag -width "NV_FLAG_IGNORE_CASE" -compact -offset indent
 .It Dv NV_FLAG_IGNORE_CASE
 Perform case-insensitive lookups of provided names.
 .El
 .Pp
 The
 .Fn nvlist_destroy
 function destroys the given nvlist.
 Function does nothing if
 .Dv NULL
 nvlist is provided.
 Function never modifies the
 .Va errno
 global variable.
 .Pp
 The
 .Fn nvlist_error
 function returns any error value that the nvlist accumulated.
 If the given nvlist is
 .Dv NULL
 the
 .Er ENOMEM
 error will be returned.
 .Pp
 The
 .Fn nvlist_empty
 functions returns
 .Dv true
 if the given nvlist is empty and
 .Dv false
 otherwise.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_clone
 functions clones the given nvlist.
 The clone shares no resources with its origin.
 This also means that all file descriptors that are part of the nvlist will be
 duplicated with the
 .Xr dup 2
 system call before placing them in the clone.
 .Pp
 The
 .Fn nvlist_dump
 dumps nvlist content for debugging purposes to the given file descriptor
 .Fa fd .
 .Pp
 The
 .Fn nvlist_fdump
 dumps nvlist content for debugging purposes to the given file stream
 .Fa fp .
 .Pp
 The
 .Fn nvlist_size
 function returns the size of the given nvlist after converting it to binary
 buffer with the
 .Fn nvlist_pack
 function.
 .Pp
 The
 .Fn nvlist_pack
 function converts the given nvlist to a binary buffer.
 The function allocates memory for the buffer, which should be freed with the
 .Xr free 3
 function.
 If the
 .Fa sizep
 argument is not
 .Dv NULL ,
 the size of the buffer will be stored there.
 The function returns
 .Dv NULL
 in case of an error (allocation failure).
 If the nvlist contains any file descriptors
 .Dv NULL
 will be returned.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_unpack
 function converts the given buffer to the nvlist.
 The function returns
 .Dv NULL
 in case of an error.
 .Pp
 The
 .Fn nvlist_send
 function sends the given nvlist over the socket given by the
 .Fa sock
 argument.
 Note that nvlist that contains file descriptors can only be send over
 .Xr unix 4
 domain sockets.
 .Pp
 The
 .Fn nvlist_recv
 function receives nvlist over the socket given by the
 .Fa sock
 argument.
 .Pp
 The
 .Fn nvlist_xfer
 function sends the given nvlist over the socket given by the
 .Fa sock
 argument and receives nvlist over the same socket.
 The given nvlist is always destroyed.
 .Pp
 The
 .Fn nvlist_next
 function iterates over the given nvlist returning names and types of subsequent
 elements.
 The
 .Fa cookiep
 argument allows the function to figure out which element should be returned
 next.
 The
 .Va *cookiep
 should be set to
 .Dv NULL
 for the first call and should not be changed later.
 Returning
 .Dv NULL
 means there are no more elements on the nvlist.
 The
 .Fa typep
 argument can be NULL.
 Elements may not be removed from the nvlist while traversing it.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_exists
 function returns
 .Dv true
 if element of the given name exists (besides of its type) or
 .Dv false
 otherwise.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_exists_type
 function returns
 .Dv true
 if element of the given name and the given type exists or
 .Dv false
 otherwise.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_exists_null ,
 .Fn nvlist_exists_bool ,
 .Fn nvlist_exists_number ,
 .Fn nvlist_exists_string ,
 .Fn nvlist_exists_nvlist ,
 .Fn nvlist_exists_descriptor ,
 .Fn nvlist_exists_binary
 functions return
 .Dv true
 if element of the given name and the given type determined by the function name
 exists or
 .Dv false
 otherwise.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_add_null ,
 .Fn nvlist_add_bool ,
 .Fn nvlist_add_number ,
 .Fn nvlist_add_string ,
 .Fn nvlist_add_stringf ,
 .Fn nvlist_add_stringv ,
 .Fn nvlist_add_nvlist ,
 .Fn nvlist_add_descriptor ,
 .Fn nvlist_add_binary
 functions add element to the given nvlist.
 When adding string or binary buffor the functions will allocate memory
 and copy the data over.
 When adding nvlist, the nvlist will be cloned and clone will be added.
 When adding descriptor, the descriptor will be duplicated using the
 .Xr dup 2
 system call and the new descriptor will be added.
 If an error occurs while adding new element, internal error is set which can be
 examined using the
 .Fn nvlist_error
 function.
 .Pp
 The
 .Fn nvlist_move_string ,
 .Fn nvlist_move_nvlist ,
 .Fn nvlist_move_descriptor ,
 .Fn nvlist_move_binary
 functions add new element to the given nvlist, but unlike
 .Fn nvlist_add_<type>
 functions they will consume the given resource.
 If an error occurs while adding new element, the resource is destroyed and
 internal error is set which can be examined using the
 .Fn nvlist_error
 function.
 .Pp
 The
 .Fn nvlist_get_bool ,
 .Fn nvlist_get_number ,
 .Fn nvlist_get_string ,
 .Fn nvlist_get_nvlist ,
 .Fn nvlist_get_descriptor ,
 .Fn nvlist_get_binary
 functions allow to obtain value of the given name.
 In case of string, nvlist, descriptor or binary, returned resource should
 not be modified - it still belongs to the nvlist.
 If element of the given name does not exist, the program will be aborted.
 To avoid that the caller should check for existence before trying to obtain
 the value or use
 .Xr dnvlist 3
 extension, which allows to provide default value for a missing element.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_get_parent
 function allows to obtain the parent nvlist from the nested nvlist.
 .Pp
 The
 .Fn nvlist_take_bool ,
 .Fn nvlist_take_number ,
 .Fn nvlist_take_string ,
 .Fn nvlist_take_nvlist ,
 .Fn nvlist_take_descriptor ,
 .Fn nvlist_take_binary
 functions return value associated with the given name and remove the element
 from the nvlist.
 In case of string and binary values, the caller is responsible for free returned
 memory using the
 .Xr free 3
 function.
 In case of nvlist, the caller is responsible for destroying returned nvlist
 using the
 .Fn nvlist_destroy
 function.
 In case of descriptor, the caller is responsible for closing returned descriptor
 using the
 .Fn close 2
 system call.
 If element of the given name does not exist, the program will be aborted.
 To avoid that the caller should check for existence before trying to obtain
 the value or use
 .Xr dnvlist 3
 extension, which allows to provide default value for a missing element.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_free
 function removes element of the given name from the nvlist (besides of its type)
 and frees all resources associated with it.
 If element of the given name does not exist, the program will be aborted.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_free_type
 function removes element of the given name and the given type from the nvlist
 and frees all resources associated with it.
 If element of the given name and the given type does not exist, the program
 will be aborted.
 The nvlist must not be in error state.
 .Pp
 The
 .Fn nvlist_free_null ,
 .Fn nvlist_free_bool ,
 .Fn nvlist_free_number ,
 .Fn nvlist_free_string ,
 .Fn nvlist_free_nvlist ,
 .Fn nvlist_free_descriptor ,
 .Fn nvlist_free_binary
 functions remove element of the given name and the given type determined by the
 function name from the nvlist and free all resources associated with it.
 If element of the given name and the given type does not exist, the program
 will be aborted.
 The nvlist must not be in error state.
 .Sh EXAMPLES
 The following example demonstrates how to prepare an nvlist and send it over
 .Xr unix 4
 domain socket.
 .Bd -literal
 nvlist_t *nvl;
 int fd;
 
 fd = open("/tmp/foo", O_RDONLY);
 if (fd < 0)
         err(1, "open(\\"/tmp/foo\\") failed");
 
 nvl = nvlist_create(0);
 /*
  * There is no need to check if nvlist_create() succeeded,
  * as the nvlist_add_<type>() functions can cope.
  * If it failed, nvlist_send() will fail.
  */
 nvlist_add_string(nvl, "filename", "/tmp/foo");
 nvlist_add_number(nvl, "flags", O_RDONLY);
 /*
  * We just want to send the descriptor, so we can give it
  * for the nvlist to consume (that's why we use nvlist_move
  * not nvlist_add).
  */
 nvlist_move_descriptor(nvl, "fd", fd);
 if (nvlist_send(sock, nvl) < 0) {
 	nvlist_destroy(nvl);
 	err(1, "nvlist_send() failed");
 }
 nvlist_destroy(nvl);
 .Ed
 .Pp
 Receiving nvlist and getting data:
 .Bd -literal
 nvlist_t *nvl;
 const char *command;
 char *filename;
 int fd;
 
 nvl = nvlist_recv(sock);
 if (nvl == NULL)
 	err(1, "nvlist_recv() failed");
 
 /* For command we take pointer to nvlist's buffer. */
 command = nvlist_get_string(nvl, "command");
 /*
  * For filename we remove it from the nvlist and take
  * ownership of the buffer.
  */
 filename = nvlist_take_string(nvl, "filename");
 /* The same for the descriptor. */
 fd = nvlist_take_descriptor(nvl, "fd");
 
 printf("command=%s filename=%s fd=%d\n", command, filename, fd);
 
 nvlist_destroy(nvl);
 free(filename);
 close(fd);
 /* command was freed by nvlist_destroy() */
 .Ed
 .Pp
 Iterating over nvlist:
 .Bd -literal
 nvlist_t *nvl;
 const char *name;
 void *cookie;
 int type;
 
 nvl = nvlist_recv(sock);
 if (nvl == NULL)
 	err(1, "nvlist_recv() failed");
 
 cookie = NULL;
 while ((name = nvlist_next(nvl, &type, &cookie)) != NULL) {
 	printf("%s=", name);
 	switch (type) {
 	case NV_TYPE_NUMBER:
 		printf("%ju", (uintmax_t)nvlist_get_number(nvl, name));
 		break;
 	case NV_TYPE_STRING:
 		printf("%s", nvlist_get_string(nvl, name));
 		break;
 	default:
 		printf("N/A");
 		break;
 	}
 	printf("\\n");
 }
+.Ed
+.Pp
+Iterating over every nested nvlist:
+.Bd -literal
+nvlist_t *nvl;
+const char *name;
+void *cookie;
+int type;
+
+nvl = nvlist_recv(sock);
+if (nvl == NULL)
+	err(1, "nvlist_recv() failed");
+
+cookie = NULL;
+do {
+	while ((name = nvlist_next(nvl, &type, &cookie)) != NULL) {
+		if (type == NV_TYPE_NVLIST) {
+			nvl = nvlist_get_nvlist(nvl, name);
+			cookie = NULL;
+		}
+	}
+} while ((nvl = nvlist_get_parent(nvl, &cookie)) != NULL);
 .Ed
 .Sh SEE ALSO
 .Xr close 2 ,
 .Xr dup 2 ,
 .Xr open 2 ,
 .Xr err 3 ,
 .Xr free 3 ,
 .Xr printf 3 ,
 .Xr unix 4
 .Sh HISTORY
 The
 .Nm libnv
 library appeared in
 .Fx 11.0 .
 .Sh AUTHORS
 .An -nosplit
 The
 .Nm libnv
 library was implemented by
 .An Pawel Jakub Dawidek Aq Mt pawel@dawidek.net
 under sponsorship from the FreeBSD Foundation.
Index: projects/clang360-import/lib/libnv/nv.h
===================================================================
--- projects/clang360-import/lib/libnv/nv.h	(revision 277944)
+++ projects/clang360-import/lib/libnv/nv.h	(revision 277945)
@@ -1,275 +1,275 @@
 /*-
  * Copyright (c) 2009-2013 The FreeBSD Foundation
  * All rights reserved.
  *
  * This software was developed by Pawel Jakub Dawidek under sponsorship from
  * the FreeBSD Foundation.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHORS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  *
  * $FreeBSD$
  */
 
 #ifndef	_NV_H_
 #define	_NV_H_
 
 #include <sys/cdefs.h>
 
 #include <stdarg.h>
 #include <stdbool.h>
 #include <stdint.h>
 #include <stdio.h>
 
 #ifndef	_NVLIST_T_DECLARED
 #define	_NVLIST_T_DECLARED
 struct nvlist;
 
 typedef struct nvlist nvlist_t;
 #endif
 
 #define	NV_NAME_MAX	2048
 
 #define	NV_TYPE_NONE			0
 
 #define	NV_TYPE_NULL			1
 #define	NV_TYPE_BOOL			2
 #define	NV_TYPE_NUMBER			3
 #define	NV_TYPE_STRING			4
 #define	NV_TYPE_NVLIST			5
 #define	NV_TYPE_DESCRIPTOR		6
 #define	NV_TYPE_BINARY			7
 
 /*
  * Perform case-insensitive lookups of provided names.
  */
 #define	NV_FLAG_IGNORE_CASE		0x01
 
 nvlist_t	*nvlist_create(int flags);
 void		 nvlist_destroy(nvlist_t *nvl);
 int		 nvlist_error(const nvlist_t *nvl);
 bool		 nvlist_empty(const nvlist_t *nvl);
 
 nvlist_t *nvlist_clone(const nvlist_t *nvl);
 
 void nvlist_dump(const nvlist_t *nvl, int fd);
 void nvlist_fdump(const nvlist_t *nvl, FILE *fp);
 
 size_t		 nvlist_size(const nvlist_t *nvl);
 void		*nvlist_pack(const nvlist_t *nvl, size_t *sizep);
 nvlist_t	*nvlist_unpack(const void *buf, size_t size);
 
 int nvlist_send(int sock, const nvlist_t *nvl);
 nvlist_t *nvlist_recv(int sock);
 nvlist_t *nvlist_xfer(int sock, nvlist_t *nvl);
 
 const char *nvlist_next(const nvlist_t *nvl, int *typep, void **cookiep);
 
-const nvlist_t *nvlist_get_parent(const nvlist_t *nvl);
+const nvlist_t *nvlist_get_parent(const nvlist_t *nvl, void **cookiep);
 
 /*
  * The nvlist_exists functions check if the given name (optionally of the given
  * type) exists on nvlist.
  */
 
 bool nvlist_exists(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_type(const nvlist_t *nvl, const char *name, int type);
 
 bool nvlist_exists_null(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_bool(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_number(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_string(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_nvlist(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_descriptor(const nvlist_t *nvl, const char *name);
 bool nvlist_exists_binary(const nvlist_t *nvl, const char *name);
 
 /*
  * The nvlist_add functions add the given name/value pair.
  * If a pointer is provided, nvlist_add will internally allocate memory for the
  * given data (in other words it won't consume provided buffer).
  */
 
 void nvlist_add_null(nvlist_t *nvl, const char *name);
 void nvlist_add_bool(nvlist_t *nvl, const char *name, bool value);
 void nvlist_add_number(nvlist_t *nvl, const char *name, uint64_t value);
 void nvlist_add_string(nvlist_t *nvl, const char *name, const char *value);
 void nvlist_add_stringf(nvlist_t *nvl, const char *name, const char *valuefmt, ...) __printflike(3, 4);
 void nvlist_add_stringv(nvlist_t *nvl, const char *name, const char *valuefmt, va_list valueap) __printflike(3, 0);
 void nvlist_add_nvlist(nvlist_t *nvl, const char *name, const nvlist_t *value);
 void nvlist_add_descriptor(nvlist_t *nvl, const char *name, int value);
 void nvlist_add_binary(nvlist_t *nvl, const char *name, const void *value, size_t size);
 
 /*
  * The nvlist_move functions add the given name/value pair.
  * The functions consumes provided buffer.
  */
 
 void nvlist_move_string(nvlist_t *nvl, const char *name, char *value);
 void nvlist_move_nvlist(nvlist_t *nvl, const char *name, nvlist_t *value);
 void nvlist_move_descriptor(nvlist_t *nvl, const char *name, int value);
 void nvlist_move_binary(nvlist_t *nvl, const char *name, void *value, size_t size);
 
 /*
  * The nvlist_get functions returns value associated with the given name.
  * If it returns a pointer, the pointer represents internal buffer and should
  * not be freed by the caller.
  */
 
 bool		 nvlist_get_bool(const nvlist_t *nvl, const char *name);
 uint64_t	 nvlist_get_number(const nvlist_t *nvl, const char *name);
 const char	*nvlist_get_string(const nvlist_t *nvl, const char *name);
 const nvlist_t	*nvlist_get_nvlist(const nvlist_t *nvl, const char *name);
 int		 nvlist_get_descriptor(const nvlist_t *nvl, const char *name);
 const void	*nvlist_get_binary(const nvlist_t *nvl, const char *name, size_t *sizep);
 
 /*
  * The nvlist_take functions returns value associated with the given name and
  * remove the given entry from the nvlist.
  * The caller is responsible for freeing received data.
  */
 
 bool		 nvlist_take_bool(nvlist_t *nvl, const char *name);
 uint64_t	 nvlist_take_number(nvlist_t *nvl, const char *name);
 char		*nvlist_take_string(nvlist_t *nvl, const char *name);
 nvlist_t	*nvlist_take_nvlist(nvlist_t *nvl, const char *name);
 int		 nvlist_take_descriptor(nvlist_t *nvl, const char *name);
 void		*nvlist_take_binary(nvlist_t *nvl, const char *name, size_t *sizep);
 
 /*
  * The nvlist_free functions removes the given name/value pair from the nvlist
  * and frees memory associated with it.
  */
 
 void nvlist_free(nvlist_t *nvl, const char *name);
 void nvlist_free_type(nvlist_t *nvl, const char *name, int type);
 
 void nvlist_free_null(nvlist_t *nvl, const char *name);
 void nvlist_free_bool(nvlist_t *nvl, const char *name);
 void nvlist_free_number(nvlist_t *nvl, const char *name);
 void nvlist_free_string(nvlist_t *nvl, const char *name);
 void nvlist_free_nvlist(nvlist_t *nvl, const char *name);
 void nvlist_free_descriptor(nvlist_t *nvl, const char *name);
 void nvlist_free_binary(nvlist_t *nvl, const char *name);
 
 /*
  * Below are the same functions, but which operate on format strings and
  * variable argument lists.
  */
 
 bool nvlist_existsf(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_type(const nvlist_t *nvl, int type, const char *namefmt, ...) __printflike(3, 4);
 
 bool nvlist_existsf_null(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_bool(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_number(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_string(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_nvlist(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_descriptor(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 bool nvlist_existsf_binary(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 
 bool nvlist_existsv(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_type(const nvlist_t *nvl, int type, const char *namefmt, va_list nameap) __printflike(3, 0);
 
 bool nvlist_existsv_null(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_bool(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_number(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_string(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_nvlist(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_descriptor(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 bool nvlist_existsv_binary(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 
 void nvlist_addf_null(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_addf_bool(nvlist_t *nvl, bool value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_addf_number(nvlist_t *nvl, uint64_t value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_addf_string(nvlist_t *nvl, const char *value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_addf_nvlist(nvlist_t *nvl, const nvlist_t *value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_addf_descriptor(nvlist_t *nvl, int value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_addf_binary(nvlist_t *nvl, const void *value, size_t size, const char *namefmt, ...) __printflike(4, 5);
 
 void nvlist_addv_null(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_addv_bool(nvlist_t *nvl, bool value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_addv_number(nvlist_t *nvl, uint64_t value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_addv_string(nvlist_t *nvl, const char *value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_addv_nvlist(nvlist_t *nvl, const nvlist_t *value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_addv_descriptor(nvlist_t *nvl, int value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_addv_binary(nvlist_t *nvl, const void *value, size_t size, const char *namefmt, va_list nameap) __printflike(4, 0);
 
 void nvlist_movef_string(nvlist_t *nvl, char *value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_movef_nvlist(nvlist_t *nvl, nvlist_t *value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_movef_descriptor(nvlist_t *nvl, int value, const char *namefmt, ...) __printflike(3, 4);
 void nvlist_movef_binary(nvlist_t *nvl, void *value, size_t size, const char *namefmt, ...) __printflike(4, 5);
 
 void nvlist_movev_string(nvlist_t *nvl, char *value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_movev_nvlist(nvlist_t *nvl, nvlist_t *value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_movev_descriptor(nvlist_t *nvl, int value, const char *namefmt, va_list nameap) __printflike(3, 0);
 void nvlist_movev_binary(nvlist_t *nvl, void *value, size_t size, const char *namefmt, va_list nameap) __printflike(4, 0);
 
 bool		 nvlist_getf_bool(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 uint64_t	 nvlist_getf_number(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 const char	*nvlist_getf_string(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 const nvlist_t	*nvlist_getf_nvlist(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 int		 nvlist_getf_descriptor(const nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 const void	*nvlist_getf_binary(const nvlist_t *nvl, size_t *sizep, const char *namefmt, ...) __printflike(3, 4);
 
 bool		 nvlist_getv_bool(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 uint64_t	 nvlist_getv_number(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 const char	*nvlist_getv_string(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 const nvlist_t	*nvlist_getv_nvlist(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 int		 nvlist_getv_descriptor(const nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 const void	*nvlist_getv_binary(const nvlist_t *nvl, size_t *sizep, const char *namefmt, va_list nameap) __printflike(3, 0);
 
 bool		 nvlist_takef_bool(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 uint64_t	 nvlist_takef_number(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 char		*nvlist_takef_string(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 nvlist_t	*nvlist_takef_nvlist(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 int		 nvlist_takef_descriptor(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void		*nvlist_takef_binary(nvlist_t *nvl, size_t *sizep, const char *namefmt, ...) __printflike(3, 4);
 
 bool		 nvlist_takev_bool(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 uint64_t	 nvlist_takev_number(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 char		*nvlist_takev_string(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 nvlist_t	*nvlist_takev_nvlist(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 int		 nvlist_takev_descriptor(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void		*nvlist_takev_binary(nvlist_t *nvl, size_t *sizep, const char *namefmt, va_list nameap) __printflike(3, 0);
 
 void nvlist_freef(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_type(nvlist_t *nvl, int type, const char *namefmt, ...) __printflike(3, 4);
 
 void nvlist_freef_null(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_bool(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_number(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_string(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_nvlist(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_descriptor(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 void nvlist_freef_binary(nvlist_t *nvl, const char *namefmt, ...) __printflike(2, 3);
 
 void nvlist_freev(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_type(nvlist_t *nvl, int type, const char *namefmt, va_list nameap) __printflike(3, 0);
 
 void nvlist_freev_null(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_bool(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_number(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_string(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_nvlist(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_descriptor(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 void nvlist_freev_binary(nvlist_t *nvl, const char *namefmt, va_list nameap) __printflike(2, 0);
 
 #endif	/* !_NV_H_ */
Index: projects/clang360-import/lib/libnv/nvlist.c
===================================================================
--- projects/clang360-import/lib/libnv/nvlist.c	(revision 277944)
+++ projects/clang360-import/lib/libnv/nvlist.c	(revision 277945)
@@ -1,1848 +1,1876 @@
 /*-
  * Copyright (c) 2009-2013 The FreeBSD Foundation
  * All rights reserved.
  *
  * This software was developed by Pawel Jakub Dawidek under sponsorship from
  * the FreeBSD Foundation.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHORS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/endian.h>
 #include <sys/queue.h>
 #include <sys/socket.h>
 
 #include <errno.h>
 #include <stdarg.h>
 #include <stdbool.h>
 #include <stdint.h>
 #define	_WITH_DPRINTF
 #include <stdio.h>
 #include <stdlib.h>
 #include <string.h>
 #include <unistd.h>
 
 #ifdef HAVE_PJDLOG
 #include <pjdlog.h>
 #endif
 
 #include "msgio.h"
 #include "nv.h"
 #include "nv_impl.h"
 #include "nvlist_impl.h"
 #include "nvpair_impl.h"
 
 #ifndef	HAVE_PJDLOG
 #include <assert.h>
 #define	PJDLOG_ASSERT(...)		assert(__VA_ARGS__)
 #define	PJDLOG_RASSERT(expr, ...)	assert(expr)
 #define	PJDLOG_ABORT(...)		do {				\
 	fprintf(stderr, "%s:%u: ", __FILE__, __LINE__);			\
 	fprintf(stderr, __VA_ARGS__);					\
 	fprintf(stderr, "\n");						\
 	abort();							\
 } while (0)
 #endif
 
 #define	NV_FLAG_PRIVATE_MASK	(NV_FLAG_BIG_ENDIAN)
 #define	NV_FLAG_PUBLIC_MASK	(NV_FLAG_IGNORE_CASE)
 #define	NV_FLAG_ALL_MASK	(NV_FLAG_PRIVATE_MASK | NV_FLAG_PUBLIC_MASK)
 
 #define	NVLIST_MAGIC	0x6e766c	/* "nvl" */
 struct nvlist {
 	int		 nvl_magic;
 	int		 nvl_error;
 	int		 nvl_flags;
 	nvpair_t	*nvl_parent;
 	struct nvl_head	 nvl_head;
 };
 
 #define	NVLIST_ASSERT(nvl)	do {					\
 	PJDLOG_ASSERT((nvl) != NULL);					\
 	PJDLOG_ASSERT((nvl)->nvl_magic == NVLIST_MAGIC);		\
 } while (0)
 
 #define	NVPAIR_ASSERT(nvp)	nvpair_assert(nvp)
 
 #define	NVLIST_HEADER_MAGIC	0x6c
 #define	NVLIST_HEADER_VERSION	0x00
 struct nvlist_header {
 	uint8_t		nvlh_magic;
 	uint8_t		nvlh_version;
 	uint8_t		nvlh_flags;
 	uint64_t	nvlh_descriptors;
 	uint64_t	nvlh_size;
 } __packed;
 
 nvlist_t *
 nvlist_create(int flags)
 {
 	nvlist_t *nvl;
 
 	PJDLOG_ASSERT((flags & ~(NV_FLAG_PUBLIC_MASK)) == 0);
 
 	nvl = malloc(sizeof(*nvl));
 	nvl->nvl_error = 0;
 	nvl->nvl_flags = flags;
 	nvl->nvl_parent = NULL;
 	TAILQ_INIT(&nvl->nvl_head);
 	nvl->nvl_magic = NVLIST_MAGIC;
 
 	return (nvl);
 }
 
 void
 nvlist_destroy(nvlist_t *nvl)
 {
 	nvpair_t *nvp;
 	int serrno;
 
 	if (nvl == NULL)
 		return;
 
 	serrno = errno;
 
 	NVLIST_ASSERT(nvl);
 
 	while ((nvp = nvlist_first_nvpair(nvl)) != NULL) {
 		nvlist_remove_nvpair(nvl, nvp);
 		nvpair_free(nvp);
 	}
 	nvl->nvl_magic = 0;
 	free(nvl);
 
 	errno = serrno;
 }
 
 int
 nvlist_error(const nvlist_t *nvl)
 {
 
 	if (nvl == NULL)
 		return (ENOMEM);
 
 	NVLIST_ASSERT(nvl);
 
 	return (nvl->nvl_error);
 }
 
 nvpair_t *
 nvlist_get_nvpair_parent(const nvlist_t *nvl)
 {
 
 	NVLIST_ASSERT(nvl);
 
 	return (nvl->nvl_parent);
 }
 
 const nvlist_t *
-nvlist_get_parent(const nvlist_t *nvl)
+nvlist_get_parent(const nvlist_t *nvl, void **cookiep)
 {
+	nvpair_t *nvp;
 
 	NVLIST_ASSERT(nvl);
 
-	if (nvl->nvl_parent == NULL)
+	nvp = nvl->nvl_parent;
+	if (cookiep != NULL)
+		*cookiep = nvp;
+	if (nvp == NULL)
 		return (NULL);
 
-	return (nvpair_nvlist(nvl->nvl_parent));
+	return (nvpair_nvlist(nvp));
 }
 
 void
 nvlist_set_parent(nvlist_t *nvl, nvpair_t *parent)
 {
 
 	NVLIST_ASSERT(nvl);
 
 	nvl->nvl_parent = parent;
 }
 
 bool
 nvlist_empty(const nvlist_t *nvl)
 {
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 
 	return (nvlist_first_nvpair(nvl) == NULL);
 }
 
 static void
 nvlist_report_missing(int type, const char *namefmt, va_list nameap)
 {
 	char *name;
 
 	vasprintf(&name, namefmt, nameap);
 	PJDLOG_ABORT("Element '%s' of type %s doesn't exist.",
 	    name != NULL ? name : "N/A", nvpair_type_string(type));
 }
 
 static nvpair_t *
 nvlist_findv(const nvlist_t *nvl, int type, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 	char *name;
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 	PJDLOG_ASSERT(type == NV_TYPE_NONE ||
 	    (type >= NV_TYPE_FIRST && type <= NV_TYPE_LAST));
 
 	if (vasprintf(&name, namefmt, nameap) < 0)
 		return (NULL);
 
 	for (nvp = nvlist_first_nvpair(nvl); nvp != NULL;
 	    nvp = nvlist_next_nvpair(nvl, nvp)) {
 		if (type != NV_TYPE_NONE && nvpair_type(nvp) != type)
 			continue;
 		if ((nvl->nvl_flags & NV_FLAG_IGNORE_CASE) != 0) {
 			if (strcasecmp(nvpair_name(nvp), name) != 0)
 				continue;
 		} else {
 			if (strcmp(nvpair_name(nvp), name) != 0)
 				continue;
 		}
 		break;
 	}
 
 	free(name);
 
 	if (nvp == NULL)
 		errno = ENOENT;
 
 	return (nvp);
 }
 
 bool
 nvlist_exists_type(const nvlist_t *nvl, const char *name, int type)
 {
 
 	return (nvlist_existsf_type(nvl, type, "%s", name));
 }
 
 bool
 nvlist_existsf_type(const nvlist_t *nvl, int type, const char *namefmt, ...)
 {
 	va_list nameap;
 	bool ret;
 
 	va_start(nameap, namefmt);
 	ret = nvlist_existsv_type(nvl, type, namefmt, nameap);
 	va_end(nameap);
 
 	return (ret);
 }
 
 bool
 nvlist_existsv_type(const nvlist_t *nvl, int type, const char *namefmt,
     va_list nameap)
 {
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 	PJDLOG_ASSERT(type == NV_TYPE_NONE ||
 	    (type >= NV_TYPE_FIRST && type <= NV_TYPE_LAST));
 
 	return (nvlist_findv(nvl, type, namefmt, nameap) != NULL);
 }
 
 void
 nvlist_free_type(nvlist_t *nvl, const char *name, int type)
 {
 
 	nvlist_freef_type(nvl, type, "%s", name);
 }
 
 void
 nvlist_freef_type(nvlist_t *nvl, int type, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_freev_type(nvl, type, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_freev_type(nvlist_t *nvl, int type, const char *namefmt, va_list nameap)
 {
 	va_list cnameap;
 	nvpair_t *nvp;
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 	PJDLOG_ASSERT(type == NV_TYPE_NONE ||
 	    (type >= NV_TYPE_FIRST && type <= NV_TYPE_LAST));
 
 	va_copy(cnameap, nameap);
 	nvp = nvlist_findv(nvl, type, namefmt, cnameap);
 	va_end(cnameap);
 	if (nvp != NULL)
 		nvlist_free_nvpair(nvl, nvp);
 	else
 		nvlist_report_missing(type, namefmt, nameap);
 }
 
 nvlist_t *
 nvlist_clone(const nvlist_t *nvl)
 {
 	nvlist_t *newnvl;
 	nvpair_t *nvp, *newnvp;
 
 	NVLIST_ASSERT(nvl);
 
 	if (nvl->nvl_error != 0) {
 		errno = nvl->nvl_error;
 		return (NULL);
 	}
 
 	newnvl = nvlist_create(nvl->nvl_flags & NV_FLAG_PUBLIC_MASK);
 	for (nvp = nvlist_first_nvpair(nvl); nvp != NULL;
 	    nvp = nvlist_next_nvpair(nvl, nvp)) {
 		newnvp = nvpair_clone(nvp);
 		if (newnvp == NULL)
 			break;
 		nvlist_move_nvpair(newnvl, newnvp);
 	}
 	if (nvp != NULL) {
 		nvlist_destroy(newnvl);
 		return (NULL);
 	}
 	return (newnvl);
 }
 
 static bool
 nvlist_dump_error_check(const nvlist_t *nvl, int fd, int level)
 {
 
 	if (nvlist_error(nvl) != 0) {
 		dprintf(fd, "%*serror: %d\n", level * 4, "",
 		    nvlist_error(nvl));
 		return (true);
 	}
 
 	return (false);
 }
 
 /*
  * Dump content of nvlist.
  */
 void
 nvlist_dump(const nvlist_t *nvl, int fd)
 {
-	nvpair_t *nvp;
+	const nvlist_t *tmpnvl;
+	nvpair_t *nvp, *tmpnvp;
+	void *cookie;
 	int level;
 
 	level = 0;
 	if (nvlist_dump_error_check(nvl, fd, level))
 		return;
 
 	nvp = nvlist_first_nvpair(nvl);
 	while (nvp != NULL) {
 		dprintf(fd, "%*s%s (%s):", level * 4, "", nvpair_name(nvp),
 		    nvpair_type_string(nvpair_type(nvp)));
 		switch (nvpair_type(nvp)) {
 		case NV_TYPE_NULL:
 			dprintf(fd, " null\n");
 			break;
 		case NV_TYPE_BOOL:
 			dprintf(fd, " %s\n", nvpair_get_bool(nvp) ?
 			    "TRUE" : "FALSE");
 			break;
 		case NV_TYPE_NUMBER:
 			dprintf(fd, " %ju (%jd) (0x%jx)\n",
 			    (uintmax_t)nvpair_get_number(nvp),
 			    (intmax_t)nvpair_get_number(nvp),
 			    (uintmax_t)nvpair_get_number(nvp));
 			break;
 		case NV_TYPE_STRING:
 			dprintf(fd, " [%s]\n", nvpair_get_string(nvp));
 			break;
 		case NV_TYPE_NVLIST:
 			dprintf(fd, "\n");
-			nvl = nvpair_get_nvlist(nvp);
-			if (nvlist_dump_error_check(nvl, fd, level + 1)) {
-				nvl = nvlist_get_parent(nvl);
+			tmpnvl = nvpair_get_nvlist(nvp);
+			if (nvlist_dump_error_check(tmpnvl, fd, level + 1))
 				break;
+			tmpnvp = nvlist_first_nvpair(tmpnvl);
+			if (tmpnvp != NULL) {
+				nvl = tmpnvl;
+				nvp = tmpnvp;
+				level++;
+				continue;
 			}
-			level += 1;
-			nvp = nvlist_first_nvpair(nvl);
-			continue;
+			break;
 		case NV_TYPE_DESCRIPTOR:
 			dprintf(fd, " %d\n", nvpair_get_descriptor(nvp));
 			break;
 		case NV_TYPE_BINARY:
 		    {
 			const unsigned char *binary;
 			unsigned int ii;
 			size_t size;
 
 			binary = nvpair_get_binary(nvp, &size);
 			dprintf(fd, " %zu ", size);
 			for (ii = 0; ii < size; ii++)
 				dprintf(fd, "%02hhx", binary[ii]);
 			dprintf(fd, "\n");
 			break;
 		    }
 		default:
 			PJDLOG_ABORT("Unknown type: %d.", nvpair_type(nvp));
 		}
 
 		while ((nvp = nvlist_next_nvpair(nvl, nvp)) == NULL) {
-			nvp = nvlist_get_nvpair_parent(nvl);
-			if (nvp == NULL)
+			cookie = NULL;
+			nvl = nvlist_get_parent(nvl, &cookie);
+			if (nvl == NULL)
 				return;
-			nvl = nvlist_get_parent(nvl);
-			level --;
+			nvp = cookie;
+			level--;
 		}
 	}
 }
 
 void
 nvlist_fdump(const nvlist_t *nvl, FILE *fp)
 {
 
 	fflush(fp);
 	nvlist_dump(nvl, fileno(fp));
 }
 
 /*
  * The function obtains size of the nvlist after nvlist_pack().
  */
 size_t
 nvlist_size(const nvlist_t *nvl)
 {
-	const nvpair_t *nvp;
+	const nvlist_t *tmpnvl;
+	const nvpair_t *nvp, *tmpnvp;
+	void *cookie;
 	size_t size;
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 
 	size = sizeof(struct nvlist_header);
 	nvp = nvlist_first_nvpair(nvl);
 	while (nvp != NULL) {
 		size += nvpair_header_size();
 		size += strlen(nvpair_name(nvp)) + 1;
 		if (nvpair_type(nvp) == NV_TYPE_NVLIST) {
 			size += sizeof(struct nvlist_header);
 			size += nvpair_header_size() + 1;
-			nvl = nvpair_get_nvlist(nvp);
-			PJDLOG_ASSERT(nvl->nvl_error == 0);
-			nvp = nvlist_first_nvpair(nvl);
-			continue;
+			tmpnvl = nvpair_get_nvlist(nvp);
+			PJDLOG_ASSERT(tmpnvl->nvl_error == 0);
+			tmpnvp = nvlist_first_nvpair(tmpnvl);
+			if (tmpnvp != NULL) {
+				nvl = tmpnvl;
+				nvp = tmpnvp;
+				continue;
+			}
 		} else {
 			size += nvpair_size(nvp);
 		}
 
 		while ((nvp = nvlist_next_nvpair(nvl, nvp)) == NULL) {
-			nvp = nvlist_get_nvpair_parent(nvl);
-			if (nvp == NULL)
+			cookie = NULL;
+			nvl = nvlist_get_parent(nvl, &cookie);
+			if (nvl == NULL)
 				goto out;
-			nvl = nvlist_get_parent(nvl);
+			nvp = cookie;
 		}
 	}
 
 out:
 	return (size);
 }
 
 static int *
 nvlist_xdescriptors(const nvlist_t *nvl, int *descs, int level)
 {
 	const nvpair_t *nvp;
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 	PJDLOG_ASSERT(level < 3);
 
 	for (nvp = nvlist_first_nvpair(nvl); nvp != NULL;
 	    nvp = nvlist_next_nvpair(nvl, nvp)) {
 		switch (nvpair_type(nvp)) {
 		case NV_TYPE_DESCRIPTOR:
 			*descs = nvpair_get_descriptor(nvp);
 			descs++;
 			break;
 		case NV_TYPE_NVLIST:
 			descs = nvlist_xdescriptors(nvpair_get_nvlist(nvp),
 			    descs, level + 1);
 			break;
 		}
 	}
 
 	return (descs);
 }
 
 int *
 nvlist_descriptors(const nvlist_t *nvl, size_t *nitemsp)
 {
 	size_t nitems;
 	int *fds;
 
 	nitems = nvlist_ndescriptors(nvl);
 	fds = malloc(sizeof(fds[0]) * (nitems + 1));
 	if (fds == NULL)
 		return (NULL);
 	if (nitems > 0)
 		nvlist_xdescriptors(nvl, fds, 0);
 	fds[nitems] = -1;
 	if (nitemsp != NULL)
 		*nitemsp = nitems;
 	return (fds);
 }
 
 static size_t
 nvlist_xndescriptors(const nvlist_t *nvl, int level)
 {
 	const nvpair_t *nvp;
 	size_t ndescs;
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(nvl->nvl_error == 0);
 	PJDLOG_ASSERT(level < 3);
 
 	ndescs = 0;
 	for (nvp = nvlist_first_nvpair(nvl); nvp != NULL;
 	    nvp = nvlist_next_nvpair(nvl, nvp)) {
 		switch (nvpair_type(nvp)) {
 		case NV_TYPE_DESCRIPTOR:
 			ndescs++;
 			break;
 		case NV_TYPE_NVLIST:
 			ndescs += nvlist_xndescriptors(nvpair_get_nvlist(nvp),
 			    level + 1);
 			break;
 		}
 	}
 
 	return (ndescs);
 }
 
 size_t
 nvlist_ndescriptors(const nvlist_t *nvl)
 {
 
 	return (nvlist_xndescriptors(nvl, 0));
 }
 
 static unsigned char *
 nvlist_pack_header(const nvlist_t *nvl, unsigned char *ptr, size_t *leftp)
 {
 	struct nvlist_header nvlhdr;
 
 	NVLIST_ASSERT(nvl);
 
 	nvlhdr.nvlh_magic = NVLIST_HEADER_MAGIC;
 	nvlhdr.nvlh_version = NVLIST_HEADER_VERSION;
 	nvlhdr.nvlh_flags = nvl->nvl_flags;
 #if BYTE_ORDER == BIG_ENDIAN
 	nvlhdr.nvlh_flags |= NV_FLAG_BIG_ENDIAN;
 #endif
 	nvlhdr.nvlh_descriptors = nvlist_ndescriptors(nvl);
 	nvlhdr.nvlh_size = *leftp - sizeof(nvlhdr);
 	PJDLOG_ASSERT(*leftp >= sizeof(nvlhdr));
 	memcpy(ptr, &nvlhdr, sizeof(nvlhdr));
 	ptr += sizeof(nvlhdr);
 	*leftp -= sizeof(nvlhdr);
 
 	return (ptr);
 }
 
 void *
 nvlist_xpack(const nvlist_t *nvl, int64_t *fdidxp, size_t *sizep)
 {
 	unsigned char *buf, *ptr;
 	size_t left, size;
-	nvpair_t *nvp;
+	const nvlist_t *tmpnvl;
+	nvpair_t *nvp, *tmpnvp;
+	void *cookie;
 
 	NVLIST_ASSERT(nvl);
 
 	if (nvl->nvl_error != 0) {
 		errno = nvl->nvl_error;
 		return (NULL);
 	}
 
 	size = nvlist_size(nvl);
 	buf = malloc(size);
 	if (buf == NULL)
 		return (NULL);
 
 	ptr = buf;
 	left = size;
 
 	ptr = nvlist_pack_header(nvl, ptr, &left);
 
 	nvp = nvlist_first_nvpair(nvl);
 	while (nvp != NULL) {
 		NVPAIR_ASSERT(nvp);
 
 		nvpair_init_datasize(nvp);
 		ptr = nvpair_pack_header(nvp, ptr, &left);
 		if (ptr == NULL) {
 			free(buf);
 			return (NULL);
 		}
 		switch (nvpair_type(nvp)) {
 		case NV_TYPE_NULL:
 			ptr = nvpair_pack_null(nvp, ptr, &left);
 			break;
 		case NV_TYPE_BOOL:
 			ptr = nvpair_pack_bool(nvp, ptr, &left);
 			break;
 		case NV_TYPE_NUMBER:
 			ptr = nvpair_pack_number(nvp, ptr, &left);
 			break;
 		case NV_TYPE_STRING:
 			ptr = nvpair_pack_string(nvp, ptr, &left);
 			break;
 		case NV_TYPE_NVLIST:
-			nvl = nvpair_get_nvlist(nvp);
-			nvp = nvlist_first_nvpair(nvl);
-			ptr = nvlist_pack_header(nvl, ptr, &left);
-			continue;
+			tmpnvl = nvpair_get_nvlist(nvp);
+			ptr = nvlist_pack_header(tmpnvl, ptr, &left);
+			if (ptr == NULL)
+				goto out;
+			tmpnvp = nvlist_first_nvpair(tmpnvl);
+			if (tmpnvp != NULL) {
+				nvl = tmpnvl;
+				nvp = tmpnvp;
+				continue;
+			}
+			ptr = nvpair_pack_nvlist_up(ptr, &left);
+			break;
 		case NV_TYPE_DESCRIPTOR:
 			ptr = nvpair_pack_descriptor(nvp, ptr, fdidxp, &left);
 			break;
 		case NV_TYPE_BINARY:
 			ptr = nvpair_pack_binary(nvp, ptr, &left);
 			break;
 		default:
 			PJDLOG_ABORT("Invalid type (%d).", nvpair_type(nvp));
 		}
 		if (ptr == NULL) {
 			free(buf);
 			return (NULL);
 		}
 		while ((nvp = nvlist_next_nvpair(nvl, nvp)) == NULL) {
-			nvp = nvlist_get_nvpair_parent(nvl);
-			if (nvp == NULL)
+			cookie = NULL;
+			nvl = nvlist_get_parent(nvl, &cookie);
+			if (nvl == NULL)
 				goto out;
+			nvp = cookie;
 			ptr = nvpair_pack_nvlist_up(ptr, &left);
 			if (ptr == NULL)
 				goto out;
-			nvl = nvlist_get_parent(nvl);
 		}
 	}
 
 out:
 	if (sizep != NULL)
 		*sizep = size;
 	return (buf);
 }
 
 void *
 nvlist_pack(const nvlist_t *nvl, size_t *sizep)
 {
 
 	NVLIST_ASSERT(nvl);
 
 	if (nvl->nvl_error != 0) {
 		errno = nvl->nvl_error;
 		return (NULL);
 	}
 
 	if (nvlist_ndescriptors(nvl) > 0) {
 		errno = EOPNOTSUPP;
 		return (NULL);
 	}
 
 	return (nvlist_xpack(nvl, NULL, sizep));
 }
 
 static bool
 nvlist_check_header(struct nvlist_header *nvlhdrp)
 {
 
 	if (nvlhdrp->nvlh_magic != NVLIST_HEADER_MAGIC) {
 		errno = EINVAL;
 		return (false);
 	}
 	if ((nvlhdrp->nvlh_flags & ~NV_FLAG_ALL_MASK) != 0) {
 		errno = EINVAL;
 		return (false);
 	}
 #if BYTE_ORDER == BIG_ENDIAN
 	if ((nvlhdrp->nvlh_flags & NV_FLAG_BIG_ENDIAN) == 0) {
 		nvlhdrp->nvlh_size = le64toh(nvlhdrp->nvlh_size);
 		nvlhdrp->nvlh_descriptors = le64toh(nvlhdrp->nvlh_descriptors);
 	}
 #else
 	if ((nvlhdrp->nvlh_flags & NV_FLAG_BIG_ENDIAN) != 0) {
 		nvlhdrp->nvlh_size = be64toh(nvlhdrp->nvlh_size);
 		nvlhdrp->nvlh_descriptors = be64toh(nvlhdrp->nvlh_descriptors);
 	}
 #endif
 	return (true);
 }
 
 const unsigned char *
 nvlist_unpack_header(nvlist_t *nvl, const unsigned char *ptr, size_t nfds,
     bool *isbep, size_t *leftp)
 {
 	struct nvlist_header nvlhdr;
 
 	if (*leftp < sizeof(nvlhdr))
 		goto failed;
 
 	memcpy(&nvlhdr, ptr, sizeof(nvlhdr));
 
 	if (!nvlist_check_header(&nvlhdr))
 		goto failed;
 
 	if (nvlhdr.nvlh_size != *leftp - sizeof(nvlhdr))
 		goto failed;
 
 	/*
 	 * nvlh_descriptors might be smaller than nfds in embedded nvlists.
 	 */
 	if (nvlhdr.nvlh_descriptors > nfds)
 		goto failed;
 
 	if ((nvlhdr.nvlh_flags & ~NV_FLAG_ALL_MASK) != 0)
 		goto failed;
 
 	nvl->nvl_flags = (nvlhdr.nvlh_flags & NV_FLAG_PUBLIC_MASK);
 
 	ptr += sizeof(nvlhdr);
 	if (isbep != NULL)
 		*isbep = (((int)nvlhdr.nvlh_flags & NV_FLAG_BIG_ENDIAN) != 0);
 	*leftp -= sizeof(nvlhdr);
 
 	return (ptr);
 failed:
 	errno = EINVAL;
 	return (NULL);
 }
 
 nvlist_t *
 nvlist_xunpack(const void *buf, size_t size, const int *fds, size_t nfds)
 {
 	const unsigned char *ptr;
 	nvlist_t *nvl, *retnvl, *tmpnvl;
 	nvpair_t *nvp;
 	size_t left;
 	bool isbe;
 
 	left = size;
 	ptr = buf;
 
 	tmpnvl = NULL;
 	nvl = retnvl = nvlist_create(0);
 	if (nvl == NULL)
 		goto failed;
 
 	ptr = nvlist_unpack_header(nvl, ptr, nfds, &isbe, &left);
 	if (ptr == NULL)
 		goto failed;
 
 	while (left > 0) {
 		ptr = nvpair_unpack(isbe, ptr, &left, &nvp);
 		if (ptr == NULL)
 			goto failed;
 		switch (nvpair_type(nvp)) {
 		case NV_TYPE_NULL:
 			ptr = nvpair_unpack_null(isbe, nvp, ptr, &left);
 			break;
 		case NV_TYPE_BOOL:
 			ptr = nvpair_unpack_bool(isbe, nvp, ptr, &left);
 			break;
 		case NV_TYPE_NUMBER:
 			ptr = nvpair_unpack_number(isbe, nvp, ptr, &left);
 			break;
 		case NV_TYPE_STRING:
 			ptr = nvpair_unpack_string(isbe, nvp, ptr, &left);
 			break;
 		case NV_TYPE_NVLIST:
 			ptr = nvpair_unpack_nvlist(isbe, nvp, ptr, &left, nfds,
 			    &tmpnvl);
 			nvlist_set_parent(tmpnvl, nvp);
 			break;
 		case NV_TYPE_DESCRIPTOR:
 			ptr = nvpair_unpack_descriptor(isbe, nvp, ptr, &left,
 			    fds, nfds);
 			break;
 		case NV_TYPE_BINARY:
 			ptr = nvpair_unpack_binary(isbe, nvp, ptr, &left);
 			break;
 		case NV_TYPE_NVLIST_UP:
 			if (nvl->nvl_parent == NULL)
 				goto failed;
 			nvl = nvpair_nvlist(nvl->nvl_parent);
 			continue;
 		default:
 			PJDLOG_ABORT("Invalid type (%d).", nvpair_type(nvp));
 		}
 		if (ptr == NULL)
 			goto failed;
 		nvlist_move_nvpair(nvl, nvp);
 		if (tmpnvl != NULL) {
 			nvl = tmpnvl;
 			tmpnvl = NULL;
 		}
 	}
 
 	return (retnvl);
 failed:
 	nvlist_destroy(retnvl);
 	return (NULL);
 }
 
 nvlist_t *
 nvlist_unpack(const void *buf, size_t size)
 {
 
 	return (nvlist_xunpack(buf, size, NULL, 0));
 }
 
 int
 nvlist_send(int sock, const nvlist_t *nvl)
 {
 	size_t datasize, nfds;
 	int *fds;
 	void *data;
 	int64_t fdidx;
 	int serrno, ret;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return (-1);
 	}
 
 	fds = nvlist_descriptors(nvl, &nfds);
 	if (fds == NULL)
 		return (-1);
 
 	ret = -1;
 	data = NULL;
 	fdidx = 0;
 
 	data = nvlist_xpack(nvl, &fdidx, &datasize);
 	if (data == NULL)
 		goto out;
 
 	if (buf_send(sock, data, datasize) == -1)
 		goto out;
 
 	if (nfds > 0) {
 		if (fd_send(sock, fds, nfds) == -1)
 			goto out;
 	}
 
 	ret = 0;
 out:
 	serrno = errno;
 	free(fds);
 	free(data);
 	errno = serrno;
 	return (ret);
 }
 
 nvlist_t *
 nvlist_recv(int sock)
 {
 	struct nvlist_header nvlhdr;
 	nvlist_t *nvl, *ret;
 	unsigned char *buf;
 	size_t nfds, size, i;
 	int serrno, *fds;
 
 	if (buf_recv(sock, &nvlhdr, sizeof(nvlhdr)) == -1)
 		return (NULL);
 
 	if (!nvlist_check_header(&nvlhdr))
 		return (NULL);
 
 	nfds = (size_t)nvlhdr.nvlh_descriptors;
 	size = sizeof(nvlhdr) + (size_t)nvlhdr.nvlh_size;
 
 	buf = malloc(size);
 	if (buf == NULL)
 		return (NULL);
 
 	memcpy(buf, &nvlhdr, sizeof(nvlhdr));
 
 	ret = NULL;
 	fds = NULL;
 
 	if (buf_recv(sock, buf + sizeof(nvlhdr), size - sizeof(nvlhdr)) == -1)
 		goto out;
 
 	if (nfds > 0) {
 		fds = malloc(nfds * sizeof(fds[0]));
 		if (fds == NULL)
 			goto out;
 		if (fd_recv(sock, fds, nfds) == -1)
 			goto out;
 	}
 
 	nvl = nvlist_xunpack(buf, size, fds, nfds);
 	if (nvl == NULL) {
 		for (i = 0; i < nfds; i++)
 			close(fds[i]);
 		goto out;
 	}
 
 	ret = nvl;
 out:
 	serrno = errno;
 	free(buf);
 	free(fds);
 	errno = serrno;
 
 	return (ret);
 }
 
 nvlist_t *
 nvlist_xfer(int sock, nvlist_t *nvl)
 {
 
 	if (nvlist_send(sock, nvl) < 0) {
 		nvlist_destroy(nvl);
 		return (NULL);
 	}
 	nvlist_destroy(nvl);
 	return (nvlist_recv(sock));
 }
 
 nvpair_t *
 nvlist_first_nvpair(const nvlist_t *nvl)
 {
 
 	NVLIST_ASSERT(nvl);
 
 	return (TAILQ_FIRST(&nvl->nvl_head));
 }
 
 nvpair_t *
 nvlist_next_nvpair(const nvlist_t *nvl, const nvpair_t *nvp)
 {
 	nvpair_t *retnvp;
 
 	NVLIST_ASSERT(nvl);
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvpair_nvlist(nvp) == nvl);
 
 	retnvp = nvpair_next(nvp);
 	PJDLOG_ASSERT(retnvp == NULL || nvpair_nvlist(retnvp) == nvl);
 
 	return (retnvp);
 
 }
 
 nvpair_t *
 nvlist_prev_nvpair(const nvlist_t *nvl, const nvpair_t *nvp)
 {
 	nvpair_t *retnvp;
 
 	NVLIST_ASSERT(nvl);
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvpair_nvlist(nvp) == nvl);
 
 	retnvp = nvpair_prev(nvp);
 	PJDLOG_ASSERT(nvpair_nvlist(retnvp) == nvl);
 
 	return (retnvp);
 }
 
 const char *
 nvlist_next(const nvlist_t *nvl, int *typep, void **cookiep)
 {
 	nvpair_t *nvp;
 
 	NVLIST_ASSERT(nvl);
 	PJDLOG_ASSERT(cookiep != NULL);
 
 	if (*cookiep == NULL)
 		nvp = nvlist_first_nvpair(nvl);
 	else
 		nvp = nvlist_next_nvpair(nvl, *cookiep);
 	if (nvp == NULL)
 		return (NULL);
 	if (typep != NULL)
 		*typep = nvpair_type(nvp);
 	*cookiep = nvp;
 	return (nvpair_name(nvp));
 }
 
 bool
 nvlist_exists(const nvlist_t *nvl, const char *name)
 {
 
 	return (nvlist_existsf(nvl, "%s", name));
 }
 
 #define	NVLIST_EXISTS(type)						\
 bool									\
 nvlist_exists_##type(const nvlist_t *nvl, const char *name)		\
 {									\
 									\
 	return (nvlist_existsf_##type(nvl, "%s", name));		\
 }
 
 NVLIST_EXISTS(null)
 NVLIST_EXISTS(bool)
 NVLIST_EXISTS(number)
 NVLIST_EXISTS(string)
 NVLIST_EXISTS(nvlist)
 NVLIST_EXISTS(descriptor)
 NVLIST_EXISTS(binary)
 
 #undef	NVLIST_EXISTS
 
 bool
 nvlist_existsf(const nvlist_t *nvl, const char *namefmt, ...)
 {
 	va_list nameap;
 	bool ret;
 
 	va_start(nameap, namefmt);
 	ret = nvlist_existsv(nvl, namefmt, nameap);
 	va_end(nameap);
 	return (ret);
 }
 
 #define	NVLIST_EXISTSF(type)						\
 bool									\
 nvlist_existsf_##type(const nvlist_t *nvl, const char *namefmt, ...)	\
 {									\
 	va_list nameap;							\
 	bool ret;							\
 									\
 	va_start(nameap, namefmt);					\
 	ret = nvlist_existsv_##type(nvl, namefmt, nameap);		\
 	va_end(nameap);							\
 	return (ret);							\
 }
 
 NVLIST_EXISTSF(null)
 NVLIST_EXISTSF(bool)
 NVLIST_EXISTSF(number)
 NVLIST_EXISTSF(string)
 NVLIST_EXISTSF(nvlist)
 NVLIST_EXISTSF(descriptor)
 NVLIST_EXISTSF(binary)
 
 #undef	NVLIST_EXISTSF
 
 bool
 nvlist_existsv(const nvlist_t *nvl, const char *namefmt, va_list nameap)
 {
 
 	return (nvlist_findv(nvl, NV_TYPE_NONE, namefmt, nameap) != NULL);
 }
 
 #define	NVLIST_EXISTSV(type, TYPE)					\
 bool									\
 nvlist_existsv_##type(const nvlist_t *nvl, const char *namefmt,		\
     va_list nameap)							\
 {									\
 									\
 	return (nvlist_findv(nvl, NV_TYPE_##TYPE, namefmt, nameap) !=	\
 	    NULL);							\
 }
 
 NVLIST_EXISTSV(null, NULL)
 NVLIST_EXISTSV(bool, BOOL)
 NVLIST_EXISTSV(number, NUMBER)
 NVLIST_EXISTSV(string, STRING)
 NVLIST_EXISTSV(nvlist, NVLIST)
 NVLIST_EXISTSV(descriptor, DESCRIPTOR)
 NVLIST_EXISTSV(binary, BINARY)
 
 #undef	NVLIST_EXISTSV
 
 void
 nvlist_add_nvpair(nvlist_t *nvl, const nvpair_t *nvp)
 {
 	nvpair_t *newnvp;
 
 	NVPAIR_ASSERT(nvp);
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 	if (nvlist_exists(nvl, nvpair_name(nvp))) {
 		nvl->nvl_error = errno = EEXIST;
 		return;
 	}
 
 	newnvp = nvpair_clone(nvp);
 	if (newnvp == NULL) {
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 		return;
 	}
 
 	nvpair_insert(&nvl->nvl_head, newnvp, nvl);
 }
 
 void
 nvlist_add_null(nvlist_t *nvl, const char *name)
 {
 
 	nvlist_addf_null(nvl, "%s", name);
 }
 
 void
 nvlist_add_bool(nvlist_t *nvl, const char *name, bool value)
 {
 
 	nvlist_addf_bool(nvl, value, "%s", name);
 }
 
 void
 nvlist_add_number(nvlist_t *nvl, const char *name, uint64_t value)
 {
 
 	nvlist_addf_number(nvl, value, "%s", name);
 }
 
 void
 nvlist_add_string(nvlist_t *nvl, const char *name, const char *value)
 {
 
 	nvlist_addf_string(nvl, value, "%s", name);
 }
 
 void
 nvlist_add_stringf(nvlist_t *nvl, const char *name, const char *valuefmt, ...)
 {
 	va_list valueap;
 
 	va_start(valueap, valuefmt);
 	nvlist_add_stringv(nvl, name, valuefmt, valueap);
 	va_end(valueap);
 }
 
 void
 nvlist_add_stringv(nvlist_t *nvl, const char *name, const char *valuefmt,
     va_list valueap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_create_stringv(name, valuefmt, valueap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_add_nvlist(nvlist_t *nvl, const char *name, const nvlist_t *value)
 {
 
 	nvlist_addf_nvlist(nvl, value, "%s", name);
 }
 
 void
 nvlist_add_descriptor(nvlist_t *nvl, const char *name, int value)
 {
 
 	nvlist_addf_descriptor(nvl, value, "%s", name);
 }
 
 void
 nvlist_add_binary(nvlist_t *nvl, const char *name, const void *value,
     size_t size)
 {
 
 	nvlist_addf_binary(nvl, value, size, "%s", name);
 }
 
 void
 nvlist_addf_null(nvlist_t *nvl, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_null(nvl, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addf_bool(nvlist_t *nvl, bool value, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_bool(nvl, value, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addf_number(nvlist_t *nvl, uint64_t value, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_number(nvl, value, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addf_string(nvlist_t *nvl, const char *value, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_string(nvl, value, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addf_nvlist(nvlist_t *nvl, const nvlist_t *value, const char *namefmt,
     ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_nvlist(nvl, value, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addf_descriptor(nvlist_t *nvl, int value, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_descriptor(nvl, value, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addf_binary(nvlist_t *nvl, const void *value, size_t size,
     const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_addv_binary(nvl, value, size, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_addv_null(nvlist_t *nvl, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_null(namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_addv_bool(nvlist_t *nvl, bool value, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_bool(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_addv_number(nvlist_t *nvl, uint64_t value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_number(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_addv_string(nvlist_t *nvl, const char *value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_string(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_addv_nvlist(nvlist_t *nvl, const nvlist_t *value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_nvlist(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_addv_descriptor(nvlist_t *nvl, int value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_descriptor(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_addv_binary(nvlist_t *nvl, const void *value, size_t size,
     const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_createv_binary(value, size, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_move_nvpair(nvlist_t *nvl, nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvpair_nvlist(nvp) == NULL);
 
 	if (nvlist_error(nvl) != 0) {
 		nvpair_free(nvp);
 		errno = nvlist_error(nvl);
 		return;
 	}
 	if (nvlist_exists(nvl, nvpair_name(nvp))) {
 		nvpair_free(nvp);
 		nvl->nvl_error = errno = EEXIST;
 		return;
 	}
 
 	nvpair_insert(&nvl->nvl_head, nvp, nvl);
 }
 
 #define	NVLIST_MOVE(vtype, type)					\
 void									\
 nvlist_move_##type(nvlist_t *nvl, const char *name, vtype value)	\
 {									\
 									\
 	nvlist_movef_##type(nvl, value, "%s", name);			\
 }
 
 NVLIST_MOVE(char *, string)
 NVLIST_MOVE(nvlist_t *, nvlist)
 NVLIST_MOVE(int, descriptor)
 
 #undef	NVLIST_MOVE
 
 void
 nvlist_move_binary(nvlist_t *nvl, const char *name, void *value, size_t size)
 {
 
 	nvlist_movef_binary(nvl, value, size, "%s", name);
 }
 
 #define	NVLIST_MOVEF(vtype, type)					\
 void									\
 nvlist_movef_##type(nvlist_t *nvl, vtype value, const char *namefmt,	\
     ...)								\
 {									\
 	va_list nameap;							\
 									\
 	va_start(nameap, namefmt);					\
 	nvlist_movev_##type(nvl, value, namefmt, nameap);		\
 	va_end(nameap);							\
 }
 
 NVLIST_MOVEF(char *, string)
 NVLIST_MOVEF(nvlist_t *, nvlist)
 NVLIST_MOVEF(int, descriptor)
 
 #undef	NVLIST_MOVEF
 
 void
 nvlist_movef_binary(nvlist_t *nvl, void *value, size_t size,
     const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_movev_binary(nvl, value, size, namefmt, nameap);
 	va_end(nameap);
 }
 
 void
 nvlist_movev_string(nvlist_t *nvl, char *value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		free(value);
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_movev_string(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_movev_nvlist(nvlist_t *nvl, nvlist_t *value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		if (value != NULL && nvlist_get_nvpair_parent(value) != NULL)
 			nvlist_destroy(value);
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_movev_nvlist(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_movev_descriptor(nvlist_t *nvl, int value, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		close(value);
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_movev_descriptor(value, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 void
 nvlist_movev_binary(nvlist_t *nvl, void *value, size_t size,
     const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (nvlist_error(nvl) != 0) {
 		free(value);
 		errno = nvlist_error(nvl);
 		return;
 	}
 
 	nvp = nvpair_movev_binary(value, size, namefmt, nameap);
 	if (nvp == NULL)
 		nvl->nvl_error = errno = (errno != 0 ? errno : ENOMEM);
 	else
 		nvlist_move_nvpair(nvl, nvp);
 }
 
 #define	NVLIST_GET(ftype, type)						\
 ftype									\
 nvlist_get_##type(const nvlist_t *nvl, const char *name)		\
 {									\
 									\
 	return (nvlist_getf_##type(nvl, "%s", name));			\
 }
 
 NVLIST_GET(const nvpair_t *, nvpair)
 NVLIST_GET(bool, bool)
 NVLIST_GET(uint64_t, number)
 NVLIST_GET(const char *, string)
 NVLIST_GET(const nvlist_t *, nvlist)
 NVLIST_GET(int, descriptor)
 
 #undef	NVLIST_GET
 
 const void *
 nvlist_get_binary(const nvlist_t *nvl, const char *name, size_t *sizep)
 {
 
 	return (nvlist_getf_binary(nvl, sizep, "%s", name));
 }
 
 #define	NVLIST_GETF(ftype, type)					\
 ftype									\
 nvlist_getf_##type(const nvlist_t *nvl, const char *namefmt, ...)	\
 {									\
 	va_list nameap;							\
 	ftype value;							\
 									\
 	va_start(nameap, namefmt);					\
 	value = nvlist_getv_##type(nvl, namefmt, nameap);		\
 	va_end(nameap);							\
 									\
 	return (value);							\
 }
 
 NVLIST_GETF(const nvpair_t *, nvpair)
 NVLIST_GETF(bool, bool)
 NVLIST_GETF(uint64_t, number)
 NVLIST_GETF(const char *, string)
 NVLIST_GETF(const nvlist_t *, nvlist)
 NVLIST_GETF(int, descriptor)
 
 #undef	NVLIST_GETF
 
 const void *
 nvlist_getf_binary(const nvlist_t *nvl, size_t *sizep, const char *namefmt, ...)
 {
 	va_list nameap;
 	const void *value;
 
 	va_start(nameap, namefmt);
 	value = nvlist_getv_binary(nvl, sizep, namefmt, nameap);
 	va_end(nameap);
 
 	return (value);
 }
 
 const nvpair_t *
 nvlist_getv_nvpair(const nvlist_t *nvl, const char *namefmt, va_list nameap)
 {
 
 	return (nvlist_findv(nvl, NV_TYPE_NONE, namefmt, nameap));
 }
 
 #define	NVLIST_GETV(ftype, type, TYPE)					\
 ftype									\
 nvlist_getv_##type(const nvlist_t *nvl, const char *namefmt,		\
     va_list nameap)							\
 {									\
 	va_list cnameap;						\
 	const nvpair_t *nvp;						\
 									\
 	va_copy(cnameap, nameap);					\
 	nvp = nvlist_findv(nvl, NV_TYPE_##TYPE, namefmt, cnameap);	\
 	va_end(cnameap);						\
 	if (nvp == NULL)						\
 		nvlist_report_missing(NV_TYPE_##TYPE, namefmt, nameap);	\
 	return (nvpair_get_##type(nvp));				\
 }
 
 NVLIST_GETV(bool, bool, BOOL)
 NVLIST_GETV(uint64_t, number, NUMBER)
 NVLIST_GETV(const char *, string, STRING)
 NVLIST_GETV(const nvlist_t *, nvlist, NVLIST)
 NVLIST_GETV(int, descriptor, DESCRIPTOR)
 
 #undef	NVLIST_GETV
 
 const void *
 nvlist_getv_binary(const nvlist_t *nvl, size_t *sizep, const char *namefmt,
     va_list nameap)
 {
 	va_list cnameap;
 	const nvpair_t *nvp;
 
 	va_copy(cnameap, nameap);
 	nvp = nvlist_findv(nvl, NV_TYPE_BINARY, namefmt, cnameap);
 	va_end(cnameap);
 	if (nvp == NULL)
 		nvlist_report_missing(NV_TYPE_BINARY, namefmt, nameap);
 
 	return (nvpair_get_binary(nvp, sizep));
 }
 
 #define	NVLIST_TAKE(ftype, type)					\
 ftype									\
 nvlist_take_##type(nvlist_t *nvl, const char *name)			\
 {									\
 									\
 	return (nvlist_takef_##type(nvl, "%s", name));			\
 }
 
 NVLIST_TAKE(nvpair_t *, nvpair)
 NVLIST_TAKE(bool, bool)
 NVLIST_TAKE(uint64_t, number)
 NVLIST_TAKE(char *, string)
 NVLIST_TAKE(nvlist_t *, nvlist)
 NVLIST_TAKE(int, descriptor)
 
 #undef	NVLIST_TAKE
 
 void *
 nvlist_take_binary(nvlist_t *nvl, const char *name, size_t *sizep)
 {
 
 	return (nvlist_takef_binary(nvl, sizep, "%s", name));
 }
 
 #define	NVLIST_TAKEF(ftype, type)					\
 ftype									\
 nvlist_takef_##type(nvlist_t *nvl, const char *namefmt, ...)		\
 {									\
 	va_list nameap;							\
 	ftype value;							\
 									\
 	va_start(nameap, namefmt);					\
 	value = nvlist_takev_##type(nvl, namefmt, nameap);		\
 	va_end(nameap);							\
 									\
 	return (value);							\
 }
 
 NVLIST_TAKEF(nvpair_t *, nvpair)
 NVLIST_TAKEF(bool, bool)
 NVLIST_TAKEF(uint64_t, number)
 NVLIST_TAKEF(char *, string)
 NVLIST_TAKEF(nvlist_t *, nvlist)
 NVLIST_TAKEF(int, descriptor)
 
 #undef	NVLIST_TAKEF
 
 void *
 nvlist_takef_binary(nvlist_t *nvl, size_t *sizep, const char *namefmt, ...)
 {
 	va_list nameap;
 	void *value;
 
 	va_start(nameap, namefmt);
 	value = nvlist_takev_binary(nvl, sizep, namefmt, nameap);
 	va_end(nameap);
 
 	return (value);
 }
 
 nvpair_t *
 nvlist_takev_nvpair(nvlist_t *nvl, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	nvp = nvlist_findv(nvl, NV_TYPE_NONE, namefmt, nameap);
 	if (nvp != NULL)
 		nvlist_remove_nvpair(nvl, nvp);
 	return (nvp);
 }
 
 #define	NVLIST_TAKEV(ftype, type, TYPE)					\
 ftype									\
 nvlist_takev_##type(nvlist_t *nvl, const char *namefmt, va_list nameap)	\
 {									\
 	va_list cnameap;						\
 	nvpair_t *nvp;							\
 	ftype value;							\
 									\
 	va_copy(cnameap, nameap);					\
 	nvp = nvlist_findv(nvl, NV_TYPE_##TYPE, namefmt, cnameap);	\
 	va_end(cnameap);						\
 	if (nvp == NULL)						\
 		nvlist_report_missing(NV_TYPE_##TYPE, namefmt, nameap);	\
 	value = (ftype)(intptr_t)nvpair_get_##type(nvp);		\
 	nvlist_remove_nvpair(nvl, nvp);					\
 	nvpair_free_structure(nvp);					\
 	return (value);							\
 }
 
 NVLIST_TAKEV(bool, bool, BOOL)
 NVLIST_TAKEV(uint64_t, number, NUMBER)
 NVLIST_TAKEV(char *, string, STRING)
 NVLIST_TAKEV(nvlist_t *, nvlist, NVLIST)
 NVLIST_TAKEV(int, descriptor, DESCRIPTOR)
 
 #undef	NVLIST_TAKEV
 
 void *
 nvlist_takev_binary(nvlist_t *nvl, size_t *sizep, const char *namefmt,
     va_list nameap)
 {
 	va_list cnameap;
 	nvpair_t *nvp;
 	void *value;
 
 	va_copy(cnameap, nameap);
 	nvp = nvlist_findv(nvl, NV_TYPE_BINARY, namefmt, cnameap);
 	va_end(cnameap);
 	if (nvp == NULL)
 		nvlist_report_missing(NV_TYPE_BINARY, namefmt, nameap);
 
 	value = (void *)(intptr_t)nvpair_get_binary(nvp, sizep);
 	nvlist_remove_nvpair(nvl, nvp);
 	nvpair_free_structure(nvp);
 	return (value);
 }
 
 void
 nvlist_remove_nvpair(nvlist_t *nvl, nvpair_t *nvp)
 {
 
 	NVLIST_ASSERT(nvl);
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvpair_nvlist(nvp) == nvl);
 
 	nvpair_remove(&nvl->nvl_head, nvp, nvl);
 }
 
 void
 nvlist_free(nvlist_t *nvl, const char *name)
 {
 
 	nvlist_freef(nvl, "%s", name);
 }
 
 #define	NVLIST_FREE(type)						\
 void									\
 nvlist_free_##type(nvlist_t *nvl, const char *name)			\
 {									\
 									\
 	nvlist_freef_##type(nvl, "%s", name);				\
 }
 
 NVLIST_FREE(null)
 NVLIST_FREE(bool)
 NVLIST_FREE(number)
 NVLIST_FREE(string)
 NVLIST_FREE(nvlist)
 NVLIST_FREE(descriptor)
 NVLIST_FREE(binary)
 
 #undef	NVLIST_FREE
 
 void
 nvlist_freef(nvlist_t *nvl, const char *namefmt, ...)
 {
 	va_list nameap;
 
 	va_start(nameap, namefmt);
 	nvlist_freev(nvl, namefmt, nameap);
 	va_end(nameap);
 }
 
 #define	NVLIST_FREEF(type)						\
 void									\
 nvlist_freef_##type(nvlist_t *nvl, const char *namefmt, ...)		\
 {									\
 	va_list nameap;							\
 									\
 	va_start(nameap, namefmt);					\
 	nvlist_freev_##type(nvl, namefmt, nameap);			\
 	va_end(nameap);							\
 }
 
 NVLIST_FREEF(null)
 NVLIST_FREEF(bool)
 NVLIST_FREEF(number)
 NVLIST_FREEF(string)
 NVLIST_FREEF(nvlist)
 NVLIST_FREEF(descriptor)
 NVLIST_FREEF(binary)
 
 #undef	NVLIST_FREEF
 
 void
 nvlist_freev(nvlist_t *nvl, const char *namefmt, va_list nameap)
 {
 
 	nvlist_freev_type(nvl, NV_TYPE_NONE, namefmt, nameap);
 }
 
 #define	NVLIST_FREEV(type, TYPE)					\
 void									\
 nvlist_freev_##type(nvlist_t *nvl, const char *namefmt, va_list nameap)	\
 {									\
 									\
 	nvlist_freev_type(nvl, NV_TYPE_##TYPE, namefmt, nameap);	\
 }
 
 NVLIST_FREEV(null, NULL)
 NVLIST_FREEV(bool, BOOL)
 NVLIST_FREEV(number, NUMBER)
 NVLIST_FREEV(string, STRING)
 NVLIST_FREEV(nvlist, NVLIST)
 NVLIST_FREEV(descriptor, DESCRIPTOR)
 NVLIST_FREEV(binary, BINARY)
 #undef	NVLIST_FREEV
 
 void
 nvlist_free_nvpair(nvlist_t *nvl, nvpair_t *nvp)
 {
 
 	NVLIST_ASSERT(nvl);
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvpair_nvlist(nvp) == nvl);
 
 	nvlist_remove_nvpair(nvl, nvp);
 	nvpair_free(nvp);
 }
Index: projects/clang360-import/lib/libnv/nvpair.c
===================================================================
--- projects/clang360-import/lib/libnv/nvpair.c	(revision 277944)
+++ projects/clang360-import/lib/libnv/nvpair.c	(revision 277945)
@@ -1,1282 +1,1304 @@
 /*-
  * Copyright (c) 2009-2013 The FreeBSD Foundation
  * All rights reserved.
  *
  * This software was developed by Pawel Jakub Dawidek under sponsorship from
  * the FreeBSD Foundation.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHORS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHORS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/endian.h>
 #include <sys/queue.h>
 
 #include <errno.h>
 #include <fcntl.h>
 #include <stdarg.h>
 #include <stdbool.h>
 #include <stdint.h>
 #include <stdlib.h>
 #include <string.h>
 #include <unistd.h>
 
 #ifdef HAVE_PJDLOG
 #include <pjdlog.h>
 #endif
 
 #include "common_impl.h"
 #include "nv.h"
 #include "nv_impl.h"
 #include "nvlist_impl.h"
 #include "nvpair_impl.h"
 
 #ifndef	HAVE_PJDLOG
 #include <assert.h>
 #define	PJDLOG_ASSERT(...)		assert(__VA_ARGS__)
 #define	PJDLOG_RASSERT(expr, ...)	assert(expr)
 #define	PJDLOG_ABORT(...)		abort()
 #endif
 
 #define	NVPAIR_MAGIC	0x6e7670	/* "nvp" */
 struct nvpair {
 	int		 nvp_magic;
 	char		*nvp_name;
 	int		 nvp_type;
 	uint64_t	 nvp_data;
 	size_t		 nvp_datasize;
 	nvlist_t	*nvp_list;
 	TAILQ_ENTRY(nvpair) nvp_next;
 };
 
 #define	NVPAIR_ASSERT(nvp)	do {					\
 	PJDLOG_ASSERT((nvp) != NULL);					\
 	PJDLOG_ASSERT((nvp)->nvp_magic == NVPAIR_MAGIC);		\
 } while (0)
 
 struct nvpair_header {
 	uint8_t		nvph_type;
 	uint16_t	nvph_namesize;
 	uint64_t	nvph_datasize;
 } __packed;
 
 
 void
 nvpair_assert(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 }
 
 nvlist_t *
 nvpair_nvlist(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	return (nvp->nvp_list);
 }
 
 nvpair_t *
 nvpair_next(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_list != NULL);
 
 	return (TAILQ_NEXT(nvp, nvp_next));
 }
 
 nvpair_t *
 nvpair_prev(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_list != NULL);
 
 	return (TAILQ_PREV(nvp, nvl_head, nvp_next));
 }
 
 void
 nvpair_insert(struct nvl_head *head, nvpair_t *nvp, nvlist_t *nvl)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_list == NULL);
 	PJDLOG_ASSERT(!nvlist_exists(nvl, nvpair_name(nvp)));
 
 	TAILQ_INSERT_TAIL(head, nvp, nvp_next);
 	nvp->nvp_list = nvl;
 }
 
 static void
 nvpair_remove_nvlist(nvpair_t *nvp)
 {
 	nvlist_t *nvl;
 
 	/* XXX: DECONST is bad, mkay? */
 	nvl = __DECONST(nvlist_t *, nvpair_get_nvlist(nvp));
 	PJDLOG_ASSERT(nvl != NULL);
 	nvlist_set_parent(nvl, NULL);
 }
 
 void
 nvpair_remove(struct nvl_head *head, nvpair_t *nvp, const nvlist_t *nvl)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_list == nvl);
 
 	if (nvpair_type(nvp) == NV_TYPE_NVLIST)
 		nvpair_remove_nvlist(nvp);
 
 	TAILQ_REMOVE(head, nvp, nvp_next);
 	nvp->nvp_list = NULL;
 }
 
 nvpair_t *
 nvpair_clone(const nvpair_t *nvp)
 {
 	nvpair_t *newnvp;
 	const char *name;
 	const void *data;
 	size_t datasize;
 
 	NVPAIR_ASSERT(nvp);
 
 	name = nvpair_name(nvp);
 
 	switch (nvpair_type(nvp)) {
 	case NV_TYPE_NULL:
 		newnvp = nvpair_create_null(name);
 		break;
 	case NV_TYPE_BOOL:
 		newnvp = nvpair_create_bool(name, nvpair_get_bool(nvp));
 		break;
 	case NV_TYPE_NUMBER:
 		newnvp = nvpair_create_number(name, nvpair_get_number(nvp));
 		break;
 	case NV_TYPE_STRING:
 		newnvp = nvpair_create_string(name, nvpair_get_string(nvp));
 		break;
 	case NV_TYPE_NVLIST:
 		newnvp = nvpair_create_nvlist(name, nvpair_get_nvlist(nvp));
 		break;
 	case NV_TYPE_DESCRIPTOR:
 		newnvp = nvpair_create_descriptor(name,
 		    nvpair_get_descriptor(nvp));
 		break;
 	case NV_TYPE_BINARY:
 		data = nvpair_get_binary(nvp, &datasize);
 		newnvp = nvpair_create_binary(name, data, datasize);
 		break;
 	default:
 		PJDLOG_ABORT("Unknown type: %d.", nvpair_type(nvp));
 	}
 
 	return (newnvp);
 }
 
 size_t
 nvpair_header_size(void)
 {
 
 	return (sizeof(struct nvpair_header));
 }
 
 size_t
 nvpair_size(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	return (nvp->nvp_datasize);
 }
 
 unsigned char *
 nvpair_pack_header(const nvpair_t *nvp, unsigned char *ptr, size_t *leftp)
 {
 	struct nvpair_header nvphdr;
 	size_t namesize;
 
 	NVPAIR_ASSERT(nvp);
 
 	nvphdr.nvph_type = nvp->nvp_type;
 	namesize = strlen(nvp->nvp_name) + 1;
 	PJDLOG_ASSERT(namesize > 0 && namesize <= UINT16_MAX);
 	nvphdr.nvph_namesize = namesize;
 	nvphdr.nvph_datasize = nvp->nvp_datasize;
 	PJDLOG_ASSERT(*leftp >= sizeof(nvphdr));
 	memcpy(ptr, &nvphdr, sizeof(nvphdr));
 	ptr += sizeof(nvphdr);
 	*leftp -= sizeof(nvphdr);
 
 	PJDLOG_ASSERT(*leftp >= namesize);
 	memcpy(ptr, nvp->nvp_name, namesize);
 	ptr += namesize;
 	*leftp -= namesize;
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_null(const nvpair_t *nvp, unsigned char *ptr,
     size_t *leftp __unused)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_NULL);
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_bool(const nvpair_t *nvp, unsigned char *ptr, size_t *leftp)
 {
 	uint8_t value;
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_BOOL);
 
 	value = (uint8_t)nvp->nvp_data;
 
 	PJDLOG_ASSERT(*leftp >= sizeof(value));
 	memcpy(ptr, &value, sizeof(value));
 	ptr += sizeof(value);
 	*leftp -= sizeof(value);
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_number(const nvpair_t *nvp, unsigned char *ptr, size_t *leftp)
 {
 	uint64_t value;
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_NUMBER);
 
 	value = (uint64_t)nvp->nvp_data;
 
 	PJDLOG_ASSERT(*leftp >= sizeof(value));
 	memcpy(ptr, &value, sizeof(value));
 	ptr += sizeof(value);
 	*leftp -= sizeof(value);
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_string(const nvpair_t *nvp, unsigned char *ptr, size_t *leftp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_STRING);
 
 	PJDLOG_ASSERT(*leftp >= nvp->nvp_datasize);
 	memcpy(ptr, (const void *)(intptr_t)nvp->nvp_data, nvp->nvp_datasize);
 	ptr += nvp->nvp_datasize;
 	*leftp -= nvp->nvp_datasize;
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_nvlist_up(unsigned char *ptr, size_t *leftp)
 {
 	struct nvpair_header nvphdr;
 	size_t namesize;
 	const char *name = "";
 
 	namesize = 1;
 	nvphdr.nvph_type = NV_TYPE_NVLIST_UP;
 	nvphdr.nvph_namesize = namesize;
 	nvphdr.nvph_datasize = 0;
 	PJDLOG_ASSERT(*leftp >= sizeof(nvphdr));
 	memcpy(ptr, &nvphdr, sizeof(nvphdr));
 	ptr += sizeof(nvphdr);
 	*leftp -= sizeof(nvphdr);
 
 	PJDLOG_ASSERT(*leftp >= namesize);
 	memcpy(ptr, name, namesize);
 	ptr += namesize;
 	*leftp -= namesize;
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_descriptor(const nvpair_t *nvp, unsigned char *ptr, int64_t *fdidxp,
     size_t *leftp)
 {
 	int64_t value;
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_DESCRIPTOR);
 
 	value = (int64_t)nvp->nvp_data;
 	if (value != -1) {
 		/*
 		 * If there is a real descriptor here, we change its number
 		 * to position in the array of descriptors send via control
 		 * message.
 		 */
 		PJDLOG_ASSERT(fdidxp != NULL);
 
 		value = *fdidxp;
 		(*fdidxp)++;
 	}
 
 	PJDLOG_ASSERT(*leftp >= sizeof(value));
 	memcpy(ptr, &value, sizeof(value));
 	ptr += sizeof(value);
 	*leftp -= sizeof(value);
 
 	return (ptr);
 }
 
 unsigned char *
 nvpair_pack_binary(const nvpair_t *nvp, unsigned char *ptr, size_t *leftp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_BINARY);
 
 	PJDLOG_ASSERT(*leftp >= nvp->nvp_datasize);
 	memcpy(ptr, (const void *)(intptr_t)nvp->nvp_data, nvp->nvp_datasize);
 	ptr += nvp->nvp_datasize;
 	*leftp -= nvp->nvp_datasize;
 
 	return (ptr);
 }
 
 void
 nvpair_init_datasize(nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	if (nvp->nvp_type == NV_TYPE_NVLIST) {
 		if (nvp->nvp_data == 0) {
 			nvp->nvp_datasize = 0;
 		} else {
 			nvp->nvp_datasize =
 			    nvlist_size((const nvlist_t *)(intptr_t)nvp->nvp_data);
 		}
 	}
 }
 
 const unsigned char *
 nvpair_unpack_header(bool isbe, nvpair_t *nvp, const unsigned char *ptr,
     size_t *leftp)
 {
 	struct nvpair_header nvphdr;
 
 	if (*leftp < sizeof(nvphdr))
 		goto failed;
 
 	memcpy(&nvphdr, ptr, sizeof(nvphdr));
 	ptr += sizeof(nvphdr);
 	*leftp -= sizeof(nvphdr);
 
 #if NV_TYPE_FIRST > 0
 	if (nvphdr.nvph_type < NV_TYPE_FIRST)
 		goto failed;
 #endif
 	if (nvphdr.nvph_type > NV_TYPE_LAST &&
 	    nvphdr.nvph_type != NV_TYPE_NVLIST_UP) {
 		goto failed;
 	}
 
 #if BYTE_ORDER == BIG_ENDIAN
 	if (!isbe) {
 		nvphdr.nvph_namesize = le16toh(nvphdr.nvph_namesize);
 		nvphdr.nvph_datasize = le64toh(nvphdr.nvph_datasize);
 	}
 #else
 	if (isbe) {
 		nvphdr.nvph_namesize = be16toh(nvphdr.nvph_namesize);
 		nvphdr.nvph_datasize = be64toh(nvphdr.nvph_datasize);
 	}
 #endif
 
 	if (nvphdr.nvph_namesize > NV_NAME_MAX)
 		goto failed;
 	if (*leftp < nvphdr.nvph_namesize)
 		goto failed;
 	if (nvphdr.nvph_namesize < 1)
 		goto failed;
 	if (strnlen((const char *)ptr, nvphdr.nvph_namesize) !=
 	    (size_t)(nvphdr.nvph_namesize - 1)) {
 		goto failed;
 	}
 
 	memcpy(nvp->nvp_name, ptr, nvphdr.nvph_namesize);
 	ptr += nvphdr.nvph_namesize;
 	*leftp -= nvphdr.nvph_namesize;
 
 	if (*leftp < nvphdr.nvph_datasize)
 		goto failed;
 
 	nvp->nvp_type = nvphdr.nvph_type;
 	nvp->nvp_data = 0;
 	nvp->nvp_datasize = nvphdr.nvph_datasize;
 
 	return (ptr);
 failed:
 	errno = EINVAL;
 	return (NULL);
 }
 
 const unsigned char *
 nvpair_unpack_null(bool isbe __unused, nvpair_t *nvp, const unsigned char *ptr,
     size_t *leftp __unused)
 {
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_NULL);
 
 	if (nvp->nvp_datasize != 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack_bool(bool isbe __unused, nvpair_t *nvp, const unsigned char *ptr,
     size_t *leftp)
 {
 	uint8_t value;
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_BOOL);
 
 	if (nvp->nvp_datasize != sizeof(value)) {
 		errno = EINVAL;
 		return (NULL);
 	}
 	if (*leftp < sizeof(value)) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	memcpy(&value, ptr, sizeof(value));
 	ptr += sizeof(value);
 	*leftp -= sizeof(value);
 
 	if (value != 0 && value != 1) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	nvp->nvp_data = (uint64_t)value;
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack_number(bool isbe, nvpair_t *nvp, const unsigned char *ptr,
      size_t *leftp)
 {
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_NUMBER);
 
 	if (nvp->nvp_datasize != sizeof(uint64_t)) {
 		errno = EINVAL;
 		return (NULL);
 	}
 	if (*leftp < sizeof(uint64_t)) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	if (isbe)
 		nvp->nvp_data = be64dec(ptr);
 	else
 		nvp->nvp_data = le64dec(ptr);
 	ptr += sizeof(uint64_t);
 	*leftp -= sizeof(uint64_t);
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack_string(bool isbe __unused, nvpair_t *nvp,
     const unsigned char *ptr, size_t *leftp)
 {
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_STRING);
 
 	if (*leftp < nvp->nvp_datasize || nvp->nvp_datasize == 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	if (strnlen((const char *)ptr, nvp->nvp_datasize) !=
 	    nvp->nvp_datasize - 1) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	nvp->nvp_data = (uint64_t)(uintptr_t)strdup((const char *)ptr);
 	if (nvp->nvp_data == 0)
 		return (NULL);
 
 	ptr += nvp->nvp_datasize;
 	*leftp -= nvp->nvp_datasize;
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack_nvlist(bool isbe __unused, nvpair_t *nvp,
     const unsigned char *ptr, size_t *leftp, size_t nfds, nvlist_t **child)
 {
 	nvlist_t *value;
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_NVLIST);
 
 	if (*leftp < nvp->nvp_datasize || nvp->nvp_datasize == 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	value = nvlist_create(0);
 	if (value == NULL)
 		return (NULL);
 
 	ptr = nvlist_unpack_header(value, ptr, nfds, NULL, leftp);
 	if (ptr == NULL)
 		return (NULL);
 
 	nvp->nvp_data = (uint64_t)(uintptr_t)value;
 	*child = value;
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack_descriptor(bool isbe, nvpair_t *nvp, const unsigned char *ptr,
     size_t *leftp, const int *fds, size_t nfds)
 {
 	int64_t idx;
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_DESCRIPTOR);
 
 	if (nvp->nvp_datasize != sizeof(idx)) {
 		errno = EINVAL;
 		return (NULL);
 	}
 	if (*leftp < sizeof(idx)) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	if (isbe)
 		idx = be64dec(ptr);
 	else
 		idx = le64dec(ptr);
 
 	if (idx < 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	if ((size_t)idx >= nfds) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	nvp->nvp_data = (uint64_t)fds[idx];
 
 	ptr += sizeof(idx);
 	*leftp -= sizeof(idx);
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack_binary(bool isbe __unused, nvpair_t *nvp,
     const unsigned char *ptr, size_t *leftp)
 {
 	void *value;
 
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_BINARY);
 
 	if (*leftp < nvp->nvp_datasize || nvp->nvp_datasize == 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	value = malloc(nvp->nvp_datasize);
 	if (value == NULL)
 		return (NULL);
 
 	memcpy(value, ptr, nvp->nvp_datasize);
 	ptr += nvp->nvp_datasize;
 	*leftp -= nvp->nvp_datasize;
 
 	nvp->nvp_data = (uint64_t)(uintptr_t)value;
 
 	return (ptr);
 }
 
 const unsigned char *
 nvpair_unpack(bool isbe, const unsigned char *ptr, size_t *leftp,
     nvpair_t **nvpp)
 {
 	nvpair_t *nvp, *tmp;
 
 	nvp = calloc(1, sizeof(*nvp) + NV_NAME_MAX);
 	if (nvp == NULL)
 		return (NULL);
 	nvp->nvp_name = (char *)(nvp + 1);
 
 	ptr = nvpair_unpack_header(isbe, nvp, ptr, leftp);
 	if (ptr == NULL)
 		goto failed;
 	tmp = realloc(nvp, sizeof(*nvp) + strlen(nvp->nvp_name) + 1);
 	if (tmp == NULL)
 		goto failed;
 	nvp = tmp;
 
 	/* Update nvp_name after realloc(). */
 	nvp->nvp_name = (char *)(nvp + 1);
 	nvp->nvp_data = 0x00;
 	nvp->nvp_magic = NVPAIR_MAGIC;
 	*nvpp = nvp;
 	return (ptr);
 failed:
 	free(nvp);
 	return (NULL);
 }
 
 int
 nvpair_type(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	return (nvp->nvp_type);
 }
 
 const char *
 nvpair_name(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	return (nvp->nvp_name);
 }
 
 static nvpair_t *
 nvpair_allocv(int type, uint64_t data, size_t datasize, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 	char *name;
 	int namelen;
 
 	PJDLOG_ASSERT(type >= NV_TYPE_FIRST && type <= NV_TYPE_LAST);
 
 	namelen = vasprintf(&name, namefmt, nameap);
 	if (namelen < 0)
 		return (NULL);
 
 	PJDLOG_ASSERT(namelen > 0);
 	if (namelen >= NV_NAME_MAX) {
 		free(name);
 		errno = ENAMETOOLONG;
 		return (NULL);
 	}
 
 	nvp = calloc(1, sizeof(*nvp) + namelen + 1);
 	if (nvp != NULL) {
 		nvp->nvp_name = (char *)(nvp + 1);
 		memcpy(nvp->nvp_name, name, namelen + 1);
 		nvp->nvp_type = type;
 		nvp->nvp_data = data;
 		nvp->nvp_datasize = datasize;
 		nvp->nvp_magic = NVPAIR_MAGIC;
 	}
 	free(name);
 
 	return (nvp);
 };
 
 nvpair_t *
 nvpair_create_null(const char *name)
 {
 
 	return (nvpair_createf_null("%s", name));
 }
 
 nvpair_t *
 nvpair_create_bool(const char *name, bool value)
 {
 
 	return (nvpair_createf_bool(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_create_number(const char *name, uint64_t value)
 {
 
 	return (nvpair_createf_number(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_create_string(const char *name, const char *value)
 {
 
 	return (nvpair_createf_string(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_create_stringf(const char *name, const char *valuefmt, ...)
 {
 	va_list valueap;
 	nvpair_t *nvp;
 
 	va_start(valueap, valuefmt);
 	nvp = nvpair_create_stringv(name, valuefmt, valueap);
 	va_end(valueap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_create_stringv(const char *name, const char *valuefmt, va_list valueap)
 {
 	nvpair_t *nvp;
 	char *str;
 	int len;
 
 	len = vasprintf(&str, valuefmt, valueap);
 	if (len < 0)
 		return (NULL);
 	nvp = nvpair_create_string(name, str);
 	if (nvp == NULL)
 		free(str);
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_create_nvlist(const char *name, const nvlist_t *value)
 {
 
 	return (nvpair_createf_nvlist(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_create_descriptor(const char *name, int value)
 {
 
 	return (nvpair_createf_descriptor(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_create_binary(const char *name, const void *value, size_t size)
 {
 
 	return (nvpair_createf_binary(value, size, "%s", name));
 }
 
 nvpair_t *
 nvpair_createf_null(const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_null(namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createf_bool(bool value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_bool(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createf_number(uint64_t value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_number(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createf_string(const char *value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_string(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createf_nvlist(const nvlist_t *value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_nvlist(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createf_descriptor(int value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_descriptor(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createf_binary(const void *value, size_t size, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_createv_binary(value, size, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createv_null(const char *namefmt, va_list nameap)
 {
 
 	return (nvpair_allocv(NV_TYPE_NULL, 0, 0, namefmt, nameap));
 }
 
 nvpair_t *
 nvpair_createv_bool(bool value, const char *namefmt, va_list nameap)
 {
 
 	return (nvpair_allocv(NV_TYPE_BOOL, value ? 1 : 0, sizeof(uint8_t),
 	    namefmt, nameap));
 }
 
 nvpair_t *
 nvpair_createv_number(uint64_t value, const char *namefmt, va_list nameap)
 {
 
 	return (nvpair_allocv(NV_TYPE_NUMBER, value, sizeof(value), namefmt,
 	    nameap));
 }
 
 nvpair_t *
 nvpair_createv_string(const char *value, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 	size_t size;
 	char *data;
 
 	if (value == NULL) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	data = strdup(value);
 	if (data == NULL)
 		return (NULL);
 	size = strlen(value) + 1;
 
 	nvp = nvpair_allocv(NV_TYPE_STRING, (uint64_t)(uintptr_t)data, size,
 	    namefmt, nameap);
 	if (nvp == NULL)
 		free(data);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createv_nvlist(const nvlist_t *value, const char *namefmt,
     va_list nameap)
 {
 	nvlist_t *nvl;
 	nvpair_t *nvp;
 
 	if (value == NULL) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	nvl = nvlist_clone(value);
 	if (nvl == NULL)
 		return (NULL);
 
 	nvp = nvpair_allocv(NV_TYPE_NVLIST, (uint64_t)(uintptr_t)nvl, 0,
 	    namefmt, nameap);
 	if (nvp == NULL)
 		nvlist_destroy(nvl);
 	else
 		nvlist_set_parent(nvl, nvp);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createv_descriptor(int value, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (value < 0 || !fd_is_valid(value)) {
 		errno = EBADF;
 		return (NULL);
 	}
 
 	value = fcntl(value, F_DUPFD_CLOEXEC, 0);
 	if (value < 0)
 		return (NULL);
 
 	nvp = nvpair_allocv(NV_TYPE_DESCRIPTOR, (uint64_t)value,
 	    sizeof(int64_t), namefmt, nameap);
 	if (nvp == NULL)
 		close(value);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_createv_binary(const void *value, size_t size, const char *namefmt,
     va_list nameap)
 {
 	nvpair_t *nvp;
 	void *data;
 
 	if (value == NULL || size == 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	data = malloc(size);
 	if (data == NULL)
 		return (NULL);
 	memcpy(data, value, size);
 
 	nvp = nvpair_allocv(NV_TYPE_BINARY, (uint64_t)(uintptr_t)data, size,
 	    namefmt, nameap);
 	if (nvp == NULL)
 		free(data);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_move_string(const char *name, char *value)
 {
 
 	return (nvpair_movef_string(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_move_nvlist(const char *name, nvlist_t *value)
 {
 
 	return (nvpair_movef_nvlist(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_move_descriptor(const char *name, int value)
 {
 
 	return (nvpair_movef_descriptor(value, "%s", name));
 }
 
 nvpair_t *
 nvpair_move_binary(const char *name, void *value, size_t size)
 {
 
 	return (nvpair_movef_binary(value, size, "%s", name));
 }
 
 nvpair_t *
 nvpair_movef_string(char *value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_movev_string(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_movef_nvlist(nvlist_t *value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_movev_nvlist(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_movef_descriptor(int value, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_movev_descriptor(value, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_movef_binary(void *value, size_t size, const char *namefmt, ...)
 {
 	va_list nameap;
 	nvpair_t *nvp;
 
 	va_start(nameap, namefmt);
 	nvp = nvpair_movev_binary(value, size, namefmt, nameap);
 	va_end(nameap);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_movev_string(char *value, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
+	int serrno;
 
 	if (value == NULL) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	nvp = nvpair_allocv(NV_TYPE_STRING, (uint64_t)(uintptr_t)value,
 	    strlen(value) + 1, namefmt, nameap);
-	if (nvp == NULL)
+	if (nvp == NULL) {
+		serrno = errno;
 		free(value);
+		errno = serrno;
+	}
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_movev_nvlist(nvlist_t *value, const char *namefmt, va_list nameap)
 {
 	nvpair_t *nvp;
 
 	if (value == NULL || nvlist_get_nvpair_parent(value) != NULL) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
 	nvp = nvpair_allocv(NV_TYPE_NVLIST, (uint64_t)(uintptr_t)value, 0,
 	    namefmt, nameap);
 	if (nvp == NULL)
 		nvlist_destroy(value);
 	else
 		nvlist_set_parent(value, nvp);
 
 	return (nvp);
 }
 
 nvpair_t *
 nvpair_movev_descriptor(int value, const char *namefmt, va_list nameap)
 {
+	nvpair_t *nvp;
+	int serrno;
 
 	if (value < 0 || !fd_is_valid(value)) {
 		errno = EBADF;
 		return (NULL);
 	}
 
-	return (nvpair_allocv(NV_TYPE_DESCRIPTOR, (uint64_t)value,
-	    sizeof(int64_t), namefmt, nameap));
+	nvp = nvpair_allocv(NV_TYPE_DESCRIPTOR, (uint64_t)value,
+	    sizeof(int64_t), namefmt, nameap);
+	if (nvp == NULL) {
+		serrno = errno;
+		close(value);
+		errno = serrno;
+	}
+
+	return (nvp);
 }
 
 nvpair_t *
 nvpair_movev_binary(void *value, size_t size, const char *namefmt,
     va_list nameap)
 {
+	nvpair_t *nvp;
+	int serrno;
 
 	if (value == NULL || size == 0) {
 		errno = EINVAL;
 		return (NULL);
 	}
 
-	return (nvpair_allocv(NV_TYPE_BINARY, (uint64_t)(uintptr_t)value, size,
-	    namefmt, nameap));
+	nvp = nvpair_allocv(NV_TYPE_BINARY, (uint64_t)(uintptr_t)value, size,
+	    namefmt, nameap);
+	if (nvp == NULL) {
+		serrno = errno;
+		free(value);
+		errno = serrno;
+	}
+
+	return (nvp);
 }
 
 bool
 nvpair_get_bool(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	return (nvp->nvp_data == 1);
 }
 
 uint64_t
 nvpair_get_number(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 
 	return (nvp->nvp_data);
 }
 
 const char *
 nvpair_get_string(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_STRING);
 
 	return ((const char *)(intptr_t)nvp->nvp_data);
 }
 
 const nvlist_t *
 nvpair_get_nvlist(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_NVLIST);
 
 	return ((const nvlist_t *)(intptr_t)nvp->nvp_data);
 }
 
 int
 nvpair_get_descriptor(const nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_DESCRIPTOR);
 
 	return ((int)nvp->nvp_data);
 }
 
 const void *
 nvpair_get_binary(const nvpair_t *nvp, size_t *sizep)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_type == NV_TYPE_BINARY);
 
 	if (sizep != NULL)
 		*sizep = nvp->nvp_datasize;
 	return ((const void *)(intptr_t)nvp->nvp_data);
 }
 
 void
 nvpair_free(nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_list == NULL);
 
 	nvp->nvp_magic = 0;
 	switch (nvp->nvp_type) {
 	case NV_TYPE_DESCRIPTOR:
 		close((int)nvp->nvp_data);
 		break;
 	case NV_TYPE_NVLIST:
 		nvlist_destroy((nvlist_t *)(intptr_t)nvp->nvp_data);
 		break;
 	case NV_TYPE_STRING:
 		free((char *)(intptr_t)nvp->nvp_data);
 		break;
 	case NV_TYPE_BINARY:
 		free((void *)(intptr_t)nvp->nvp_data);
 		break;
 	}
 	free(nvp);
 }
 
 void
 nvpair_free_structure(nvpair_t *nvp)
 {
 
 	NVPAIR_ASSERT(nvp);
 	PJDLOG_ASSERT(nvp->nvp_list == NULL);
 
 	nvp->nvp_magic = 0;
 	free(nvp);
 }
 
 const char *
 nvpair_type_string(int type)
 {
 
 	switch (type) {
 	case NV_TYPE_NULL:
 		return ("NULL");
 	case NV_TYPE_BOOL:
 		return ("BOOL");
 	case NV_TYPE_NUMBER:
 		return ("NUMBER");
 	case NV_TYPE_STRING:
 		return ("STRING");
 	case NV_TYPE_NVLIST:
 		return ("NVLIST");
 	case NV_TYPE_DESCRIPTOR:
 		return ("DESCRIPTOR");
 	case NV_TYPE_BINARY:
 		return ("BINARY");
 	default:
 		return ("<UNKNOWN>");
 	}
 }
Index: projects/clang360-import/libexec/rtld-elf/rtld.c
===================================================================
--- projects/clang360-import/libexec/rtld-elf/rtld.c	(revision 277944)
+++ projects/clang360-import/libexec/rtld-elf/rtld.c	(revision 277945)
@@ -1,5056 +1,5054 @@
 /*-
  * Copyright 1996, 1997, 1998, 1999, 2000 John D. Polstra.
  * Copyright 2003 Alexander Kabaev <kan@FreeBSD.ORG>.
  * Copyright 2009-2012 Konstantin Belousov <kib@FreeBSD.ORG>.
  * Copyright 2012 John Marino <draco@marino.st>.
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
  * IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY DIRECT, INDIRECT,
  * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT
  * NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
  * DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
  * THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
  * (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF
  * THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  *
  * $FreeBSD$
  */
 
 /*
  * Dynamic linker for ELF.
  *
  * John Polstra <jdp@polstra.com>.
  */
 
 #ifndef __GNUC__
 #error "GCC is needed to compile this file"
 #endif
 
 #include <sys/param.h>
 #include <sys/mount.h>
 #include <sys/mman.h>
 #include <sys/stat.h>
 #include <sys/sysctl.h>
 #include <sys/uio.h>
 #include <sys/utsname.h>
 #include <sys/ktrace.h>
 
 #include <dlfcn.h>
 #include <err.h>
 #include <errno.h>
 #include <fcntl.h>
 #include <stdarg.h>
 #include <stdio.h>
 #include <stdlib.h>
 #include <string.h>
 #include <unistd.h>
 
 #include "debug.h"
 #include "rtld.h"
 #include "libmap.h"
 #include "rtld_tls.h"
 #include "rtld_printf.h"
 #include "notes.h"
 
 #ifndef COMPAT_32BIT
 #define PATH_RTLD	"/libexec/ld-elf.so.1"
 #else
 #define PATH_RTLD	"/libexec/ld-elf32.so.1"
 #endif
 
 /* Types. */
 typedef void (*func_ptr_type)();
 typedef void * (*path_enum_proc) (const char *path, size_t len, void *arg);
 
 /*
  * Function declarations.
  */
 static const char *basename(const char *);
 static void die(void) __dead2;
 static void digest_dynamic1(Obj_Entry *, int, const Elf_Dyn **,
     const Elf_Dyn **, const Elf_Dyn **);
 static void digest_dynamic2(Obj_Entry *, const Elf_Dyn *, const Elf_Dyn *,
     const Elf_Dyn *);
 static void digest_dynamic(Obj_Entry *, int);
 static Obj_Entry *digest_phdr(const Elf_Phdr *, int, caddr_t, const char *);
 static Obj_Entry *dlcheck(void *);
 static Obj_Entry *dlopen_object(const char *name, int fd, Obj_Entry *refobj,
     int lo_flags, int mode, RtldLockState *lockstate);
 static Obj_Entry *do_load_object(int, const char *, char *, struct stat *, int);
 static int do_search_info(const Obj_Entry *obj, int, struct dl_serinfo *);
 static bool donelist_check(DoneList *, const Obj_Entry *);
 static void errmsg_restore(char *);
 static char *errmsg_save(void);
 static void *fill_search_info(const char *, size_t, void *);
 static char *find_library(const char *, const Obj_Entry *, int *);
 static const char *gethints(bool);
 static void init_dag(Obj_Entry *);
 static void init_pagesizes(Elf_Auxinfo **aux_info);
 static void init_rtld(caddr_t, Elf_Auxinfo **);
 static void initlist_add_neededs(Needed_Entry *, Objlist *);
 static void initlist_add_objects(Obj_Entry *, Obj_Entry **, Objlist *);
 static void linkmap_add(Obj_Entry *);
 static void linkmap_delete(Obj_Entry *);
 static void load_filtees(Obj_Entry *, int flags, RtldLockState *);
 static void unload_filtees(Obj_Entry *);
 static int load_needed_objects(Obj_Entry *, int);
 static int load_preload_objects(void);
 static Obj_Entry *load_object(const char *, int fd, const Obj_Entry *, int);
 static void map_stacks_exec(RtldLockState *);
 static Obj_Entry *obj_from_addr(const void *);
 static void objlist_call_fini(Objlist *, Obj_Entry *, RtldLockState *);
 static void objlist_call_init(Objlist *, RtldLockState *);
 static void objlist_clear(Objlist *);
 static Objlist_Entry *objlist_find(Objlist *, const Obj_Entry *);
 static void objlist_init(Objlist *);
 static void objlist_push_head(Objlist *, Obj_Entry *);
 static void objlist_push_tail(Objlist *, Obj_Entry *);
 static void objlist_put_after(Objlist *, Obj_Entry *, Obj_Entry *);
 static void objlist_remove(Objlist *, Obj_Entry *);
 static int parse_libdir(const char *);
 static void *path_enumerate(const char *, path_enum_proc, void *);
 static int relocate_object_dag(Obj_Entry *root, bool bind_now,
     Obj_Entry *rtldobj, int flags, RtldLockState *lockstate);
 static int relocate_object(Obj_Entry *obj, bool bind_now, Obj_Entry *rtldobj,
     int flags, RtldLockState *lockstate);
 static int relocate_objects(Obj_Entry *, bool, Obj_Entry *, int,
     RtldLockState *);
 static int resolve_objects_ifunc(Obj_Entry *first, bool bind_now,
     int flags, RtldLockState *lockstate);
 static int rtld_dirname(const char *, char *);
 static int rtld_dirname_abs(const char *, char *);
 static void *rtld_dlopen(const char *name, int fd, int mode);
 static void rtld_exit(void);
 static char *search_library_path(const char *, const char *);
 static char *search_library_pathfds(const char *, const char *, int *);
 static const void **get_program_var_addr(const char *, RtldLockState *);
 static void set_program_var(const char *, const void *);
 static int symlook_default(SymLook *, const Obj_Entry *refobj);
 static int symlook_global(SymLook *, DoneList *);
 static void symlook_init_from_req(SymLook *, const SymLook *);
 static int symlook_list(SymLook *, const Objlist *, DoneList *);
 static int symlook_needed(SymLook *, const Needed_Entry *, DoneList *);
 static int symlook_obj1_sysv(SymLook *, const Obj_Entry *);
 static int symlook_obj1_gnu(SymLook *, const Obj_Entry *);
 static void trace_loaded_objects(Obj_Entry *);
 static void unlink_object(Obj_Entry *);
 static void unload_object(Obj_Entry *);
 static void unref_dag(Obj_Entry *);
 static void ref_dag(Obj_Entry *);
 static char *origin_subst_one(char *, const char *, const char *, bool);
 static char *origin_subst(char *, const char *);
 static void preinit_main(void);
 static int  rtld_verify_versions(const Objlist *);
 static int  rtld_verify_object_versions(Obj_Entry *);
 static void object_add_name(Obj_Entry *, const char *);
 static int  object_match_name(const Obj_Entry *, const char *);
 static void ld_utrace_log(int, void *, void *, size_t, int, const char *);
 static void rtld_fill_dl_phdr_info(const Obj_Entry *obj,
     struct dl_phdr_info *phdr_info);
 static uint32_t gnu_hash(const char *);
 static bool matched_symbol(SymLook *, const Obj_Entry *, Sym_Match_Result *,
     const unsigned long);
 
 void r_debug_state(struct r_debug *, struct link_map *) __noinline;
 void _r_debug_postinit(struct link_map *) __noinline;
 
 int __sys_openat(int, const char *, int, ...);
 
 /*
  * Data declarations.
  */
 static char *error_message;	/* Message for dlerror(), or NULL */
 struct r_debug r_debug;		/* for GDB; */
 static bool libmap_disable;	/* Disable libmap */
 static bool ld_loadfltr;	/* Immediate filters processing */
 static char *libmap_override;	/* Maps to use in addition to libmap.conf */
 static bool trust;		/* False for setuid and setgid programs */
 static bool dangerous_ld_env;	/* True if environment variables have been
 				   used to affect the libraries loaded */
 static char *ld_bind_now;	/* Environment variable for immediate binding */
 static char *ld_debug;		/* Environment variable for debugging */
 static char *ld_library_path;	/* Environment variable for search path */
 static char *ld_library_dirs;	/* Environment variable for library descriptors */
 static char *ld_preload;	/* Environment variable for libraries to
 				   load first */
 static char *ld_elf_hints_path;	/* Environment variable for alternative hints path */
 static char *ld_tracing;	/* Called from ldd to print libs */
 static char *ld_utrace;		/* Use utrace() to log events. */
 static Obj_Entry *obj_list;	/* Head of linked list of shared objects */
 static Obj_Entry **obj_tail;	/* Link field of last object in list */
 static Obj_Entry *obj_main;	/* The main program shared object */
 static Obj_Entry obj_rtld;	/* The dynamic linker shared object */
 static unsigned int obj_count;	/* Number of objects in obj_list */
 static unsigned int obj_loads;	/* Number of objects in obj_list */
 
 static Objlist list_global =	/* Objects dlopened with RTLD_GLOBAL */
   STAILQ_HEAD_INITIALIZER(list_global);
 static Objlist list_main =	/* Objects loaded at program startup */
   STAILQ_HEAD_INITIALIZER(list_main);
 static Objlist list_fini =	/* Objects needing fini() calls */
   STAILQ_HEAD_INITIALIZER(list_fini);
 
 Elf_Sym sym_zero;		/* For resolving undefined weak refs. */
 
 #define GDB_STATE(s,m)	r_debug.r_state = s; r_debug_state(&r_debug,m);
 
 extern Elf_Dyn _DYNAMIC;
 #pragma weak _DYNAMIC
 #ifndef RTLD_IS_DYNAMIC
 #define	RTLD_IS_DYNAMIC()	(&_DYNAMIC != NULL)
 #endif
 
 int npagesizes, osreldate;
 size_t *pagesizes;
 
 long __stack_chk_guard[8] = {0, 0, 0, 0, 0, 0, 0, 0};
 
 static int stack_prot = PROT_READ | PROT_WRITE | RTLD_DEFAULT_STACK_EXEC;
 static int max_stack_flags;
 
 /*
  * Global declarations normally provided by crt1.  The dynamic linker is
  * not built with crt1, so we have to provide them ourselves.
  */
 char *__progname;
 char **environ;
 
 /*
  * Used to pass argc, argv to init functions.
  */
 int main_argc;
 char **main_argv;
 
 /*
  * Globals to control TLS allocation.
  */
 size_t tls_last_offset;		/* Static TLS offset of last module */
 size_t tls_last_size;		/* Static TLS size of last module */
 size_t tls_static_space;	/* Static TLS space allocated */
 size_t tls_static_max_align;
 int tls_dtv_generation = 1;	/* Used to detect when dtv size changes  */
 int tls_max_index = 1;		/* Largest module index allocated */
 
 bool ld_library_path_rpath = false;
 
 /*
  * Fill in a DoneList with an allocation large enough to hold all of
  * the currently-loaded objects.  Keep this as a macro since it calls
  * alloca and we want that to occur within the scope of the caller.
  */
 #define donelist_init(dlp)					\
     ((dlp)->objs = alloca(obj_count * sizeof (dlp)->objs[0]),	\
     assert((dlp)->objs != NULL),				\
     (dlp)->num_alloc = obj_count,				\
     (dlp)->num_used = 0)
 
 #define	UTRACE_DLOPEN_START		1
 #define	UTRACE_DLOPEN_STOP		2
 #define	UTRACE_DLCLOSE_START		3
 #define	UTRACE_DLCLOSE_STOP		4
 #define	UTRACE_LOAD_OBJECT		5
 #define	UTRACE_UNLOAD_OBJECT		6
 #define	UTRACE_ADD_RUNDEP		7
 #define	UTRACE_PRELOAD_FINISHED		8
 #define	UTRACE_INIT_CALL		9
 #define	UTRACE_FINI_CALL		10
 #define	UTRACE_DLSYM_START		11
 #define	UTRACE_DLSYM_STOP		12
 
 struct utrace_rtld {
 	char sig[4];			/* 'RTLD' */
 	int event;
 	void *handle;
 	void *mapbase;			/* Used for 'parent' and 'init/fini' */
 	size_t mapsize;
 	int refcnt;			/* Used for 'mode' */
 	char name[MAXPATHLEN];
 };
 
 #define	LD_UTRACE(e, h, mb, ms, r, n) do {			\
 	if (ld_utrace != NULL)					\
 		ld_utrace_log(e, h, mb, ms, r, n);		\
 } while (0)
 
 static void
 ld_utrace_log(int event, void *handle, void *mapbase, size_t mapsize,
     int refcnt, const char *name)
 {
 	struct utrace_rtld ut;
 
 	ut.sig[0] = 'R';
 	ut.sig[1] = 'T';
 	ut.sig[2] = 'L';
 	ut.sig[3] = 'D';
 	ut.event = event;
 	ut.handle = handle;
 	ut.mapbase = mapbase;
 	ut.mapsize = mapsize;
 	ut.refcnt = refcnt;
 	bzero(ut.name, sizeof(ut.name));
 	if (name)
 		strlcpy(ut.name, name, sizeof(ut.name));
 	utrace(&ut, sizeof(ut));
 }
 
 /*
  * Main entry point for dynamic linking.  The first argument is the
  * stack pointer.  The stack is expected to be laid out as described
  * in the SVR4 ABI specification, Intel 386 Processor Supplement.
  * Specifically, the stack pointer points to a word containing
  * ARGC.  Following that in the stack is a null-terminated sequence
  * of pointers to argument strings.  Then comes a null-terminated
  * sequence of pointers to environment strings.  Finally, there is a
  * sequence of "auxiliary vector" entries.
  *
  * The second argument points to a place to store the dynamic linker's
  * exit procedure pointer and the third to a place to store the main
  * program's object.
  *
  * The return value is the main program's entry point.
  */
 func_ptr_type
 _rtld(Elf_Addr *sp, func_ptr_type *exit_proc, Obj_Entry **objp)
 {
     Elf_Auxinfo *aux_info[AT_COUNT];
     int i;
     int argc;
     char **argv;
     char **env;
     Elf_Auxinfo *aux;
     Elf_Auxinfo *auxp;
     const char *argv0;
     Objlist_Entry *entry;
     Obj_Entry *obj;
     Obj_Entry **preload_tail;
     Obj_Entry *last_interposer;
     Objlist initlist;
     RtldLockState lockstate;
     char *library_path_rpath;
     int mib[2];
     size_t len;
 
     /*
      * On entry, the dynamic linker itself has not been relocated yet.
      * Be very careful not to reference any global data until after
      * init_rtld has returned.  It is OK to reference file-scope statics
      * and string constants, and to call static and global functions.
      */
 
     /* Find the auxiliary vector on the stack. */
     argc = *sp++;
     argv = (char **) sp;
     sp += argc + 1;	/* Skip over arguments and NULL terminator */
     env = (char **) sp;
     while (*sp++ != 0)	/* Skip over environment, and NULL terminator */
 	;
     aux = (Elf_Auxinfo *) sp;
 
     /* Digest the auxiliary vector. */
     for (i = 0;  i < AT_COUNT;  i++)
 	aux_info[i] = NULL;
     for (auxp = aux;  auxp->a_type != AT_NULL;  auxp++) {
 	if (auxp->a_type < AT_COUNT)
 	    aux_info[auxp->a_type] = auxp;
     }
 
     /* Initialize and relocate ourselves. */
     assert(aux_info[AT_BASE] != NULL);
     init_rtld((caddr_t) aux_info[AT_BASE]->a_un.a_ptr, aux_info);
 
     __progname = obj_rtld.path;
     argv0 = argv[0] != NULL ? argv[0] : "(null)";
     environ = env;
     main_argc = argc;
     main_argv = argv;
 
     if (aux_info[AT_CANARY] != NULL &&
 	aux_info[AT_CANARY]->a_un.a_ptr != NULL) {
 	    i = aux_info[AT_CANARYLEN]->a_un.a_val;
 	    if (i > sizeof(__stack_chk_guard))
 		    i = sizeof(__stack_chk_guard);
 	    memcpy(__stack_chk_guard, aux_info[AT_CANARY]->a_un.a_ptr, i);
     } else {
 	mib[0] = CTL_KERN;
 	mib[1] = KERN_ARND;
 
 	len = sizeof(__stack_chk_guard);
 	if (sysctl(mib, 2, __stack_chk_guard, &len, NULL, 0) == -1 ||
 	    len != sizeof(__stack_chk_guard)) {
 		/* If sysctl was unsuccessful, use the "terminator canary". */
 		((unsigned char *)(void *)__stack_chk_guard)[0] = 0;
 		((unsigned char *)(void *)__stack_chk_guard)[1] = 0;
 		((unsigned char *)(void *)__stack_chk_guard)[2] = '\n';
 		((unsigned char *)(void *)__stack_chk_guard)[3] = 255;
 	}
     }
 
     trust = !issetugid();
 
     ld_bind_now = getenv(LD_ "BIND_NOW");
     /* 
      * If the process is tainted, then we un-set the dangerous environment
      * variables.  The process will be marked as tainted until setuid(2)
      * is called.  If any child process calls setuid(2) we do not want any
      * future processes to honor the potentially un-safe variables.
      */
     if (!trust) {
         if (unsetenv(LD_ "PRELOAD") || unsetenv(LD_ "LIBMAP") ||
 	    unsetenv(LD_ "LIBRARY_PATH") || unsetenv(LD_ "LIBRARY_PATH_FDS") ||
 	    unsetenv(LD_ "LIBMAP_DISABLE") ||
 	    unsetenv(LD_ "DEBUG") || unsetenv(LD_ "ELF_HINTS_PATH") ||
 	    unsetenv(LD_ "LOADFLTR") || unsetenv(LD_ "LIBRARY_PATH_RPATH")) {
 		_rtld_error("environment corrupt; aborting");
 		die();
 	}
     }
     ld_debug = getenv(LD_ "DEBUG");
     libmap_disable = getenv(LD_ "LIBMAP_DISABLE") != NULL;
     libmap_override = getenv(LD_ "LIBMAP");
     ld_library_path = getenv(LD_ "LIBRARY_PATH");
     ld_library_dirs = getenv(LD_ "LIBRARY_PATH_FDS");
     ld_preload = getenv(LD_ "PRELOAD");
     ld_elf_hints_path = getenv(LD_ "ELF_HINTS_PATH");
     ld_loadfltr = getenv(LD_ "LOADFLTR") != NULL;
     library_path_rpath = getenv(LD_ "LIBRARY_PATH_RPATH");
     if (library_path_rpath != NULL) {
 	    if (library_path_rpath[0] == 'y' ||
 		library_path_rpath[0] == 'Y' ||
 		library_path_rpath[0] == '1')
 		    ld_library_path_rpath = true;
 	    else
 		    ld_library_path_rpath = false;
     }
     dangerous_ld_env = libmap_disable || (libmap_override != NULL) ||
 	(ld_library_path != NULL) || (ld_preload != NULL) ||
 	(ld_elf_hints_path != NULL) || ld_loadfltr;
     ld_tracing = getenv(LD_ "TRACE_LOADED_OBJECTS");
     ld_utrace = getenv(LD_ "UTRACE");
 
     if ((ld_elf_hints_path == NULL) || strlen(ld_elf_hints_path) == 0)
 	ld_elf_hints_path = _PATH_ELF_HINTS;
 
     if (ld_debug != NULL && *ld_debug != '\0')
 	debug = 1;
     dbg("%s is initialized, base address = %p", __progname,
 	(caddr_t) aux_info[AT_BASE]->a_un.a_ptr);
     dbg("RTLD dynamic = %p", obj_rtld.dynamic);
     dbg("RTLD pltgot  = %p", obj_rtld.pltgot);
 
     dbg("initializing thread locks");
     lockdflt_init();
 
     /*
      * Load the main program, or process its program header if it is
      * already loaded.
      */
     if (aux_info[AT_EXECFD] != NULL) {	/* Load the main program. */
 	int fd = aux_info[AT_EXECFD]->a_un.a_val;
 	dbg("loading main program");
 	obj_main = map_object(fd, argv0, NULL);
 	close(fd);
 	if (obj_main == NULL)
 	    die();
 	max_stack_flags = obj->stack_flags;
     } else {				/* Main program already loaded. */
 	const Elf_Phdr *phdr;
 	int phnum;
 	caddr_t entry;
 
 	dbg("processing main program's program header");
 	assert(aux_info[AT_PHDR] != NULL);
 	phdr = (const Elf_Phdr *) aux_info[AT_PHDR]->a_un.a_ptr;
 	assert(aux_info[AT_PHNUM] != NULL);
 	phnum = aux_info[AT_PHNUM]->a_un.a_val;
 	assert(aux_info[AT_PHENT] != NULL);
 	assert(aux_info[AT_PHENT]->a_un.a_val == sizeof(Elf_Phdr));
 	assert(aux_info[AT_ENTRY] != NULL);
 	entry = (caddr_t) aux_info[AT_ENTRY]->a_un.a_ptr;
 	if ((obj_main = digest_phdr(phdr, phnum, entry, argv0)) == NULL)
 	    die();
     }
 
     if (aux_info[AT_EXECPATH] != 0) {
 	    char *kexecpath;
 	    char buf[MAXPATHLEN];
 
 	    kexecpath = aux_info[AT_EXECPATH]->a_un.a_ptr;
 	    dbg("AT_EXECPATH %p %s", kexecpath, kexecpath);
 	    if (kexecpath[0] == '/')
 		    obj_main->path = kexecpath;
 	    else if (getcwd(buf, sizeof(buf)) == NULL ||
 		     strlcat(buf, "/", sizeof(buf)) >= sizeof(buf) ||
 		     strlcat(buf, kexecpath, sizeof(buf)) >= sizeof(buf))
 		    obj_main->path = xstrdup(argv0);
 	    else
 		    obj_main->path = xstrdup(buf);
     } else {
 	    dbg("No AT_EXECPATH");
 	    obj_main->path = xstrdup(argv0);
     }
     dbg("obj_main path %s", obj_main->path);
     obj_main->mainprog = true;
 
     if (aux_info[AT_STACKPROT] != NULL &&
       aux_info[AT_STACKPROT]->a_un.a_val != 0)
 	    stack_prot = aux_info[AT_STACKPROT]->a_un.a_val;
 
 #ifndef COMPAT_32BIT
     /*
      * Get the actual dynamic linker pathname from the executable if
      * possible.  (It should always be possible.)  That ensures that
      * gdb will find the right dynamic linker even if a non-standard
      * one is being used.
      */
     if (obj_main->interp != NULL &&
       strcmp(obj_main->interp, obj_rtld.path) != 0) {
 	free(obj_rtld.path);
 	obj_rtld.path = xstrdup(obj_main->interp);
         __progname = obj_rtld.path;
     }
 #endif
 
     digest_dynamic(obj_main, 0);
     dbg("%s valid_hash_sysv %d valid_hash_gnu %d dynsymcount %d",
 	obj_main->path, obj_main->valid_hash_sysv, obj_main->valid_hash_gnu,
 	obj_main->dynsymcount);
 
     linkmap_add(obj_main);
     linkmap_add(&obj_rtld);
 
     /* Link the main program into the list of objects. */
     *obj_tail = obj_main;
     obj_tail = &obj_main->next;
     obj_count++;
     obj_loads++;
 
     /* Initialize a fake symbol for resolving undefined weak references. */
     sym_zero.st_info = ELF_ST_INFO(STB_GLOBAL, STT_NOTYPE);
     sym_zero.st_shndx = SHN_UNDEF;
     sym_zero.st_value = -(uintptr_t)obj_main->relocbase;
 
     if (!libmap_disable)
         libmap_disable = (bool)lm_init(libmap_override);
 
     dbg("loading LD_PRELOAD libraries");
     if (load_preload_objects() == -1)
 	die();
     preload_tail = obj_tail;
 
     dbg("loading needed objects");
     if (load_needed_objects(obj_main, 0) == -1)
 	die();
 
     /* Make a list of all objects loaded at startup. */
     last_interposer = obj_main;
     for (obj = obj_list;  obj != NULL;  obj = obj->next) {
 	if (obj->z_interpose && obj != obj_main) {
 	    objlist_put_after(&list_main, last_interposer, obj);
 	    last_interposer = obj;
 	} else {
 	    objlist_push_tail(&list_main, obj);
 	}
     	obj->refcount++;
     }
 
     dbg("checking for required versions");
     if (rtld_verify_versions(&list_main) == -1 && !ld_tracing)
 	die();
 
     if (ld_tracing) {		/* We're done */
 	trace_loaded_objects(obj_main);
 	exit(0);
     }
 
     if (getenv(LD_ "DUMP_REL_PRE") != NULL) {
        dump_relocations(obj_main);
        exit (0);
     }
 
     /*
      * Processing tls relocations requires having the tls offsets
      * initialized.  Prepare offsets before starting initial
      * relocation processing.
      */
     dbg("initializing initial thread local storage offsets");
     STAILQ_FOREACH(entry, &list_main, link) {
 	/*
 	 * Allocate all the initial objects out of the static TLS
 	 * block even if they didn't ask for it.
 	 */
 	allocate_tls_offset(entry->obj);
     }
 
     if (relocate_objects(obj_main,
       ld_bind_now != NULL && *ld_bind_now != '\0',
       &obj_rtld, SYMLOOK_EARLY, NULL) == -1)
 	die();
 
     dbg("doing copy relocations");
     if (do_copy_relocations(obj_main) == -1)
 	die();
 
     if (getenv(LD_ "DUMP_REL_POST") != NULL) {
        dump_relocations(obj_main);
        exit (0);
     }
 
     /*
      * Setup TLS for main thread.  This must be done after the
      * relocations are processed, since tls initialization section
      * might be the subject for relocations.
      */
     dbg("initializing initial thread local storage");
     allocate_initial_tls(obj_list);
 
     dbg("initializing key program variables");
     set_program_var("__progname", argv[0] != NULL ? basename(argv[0]) : "");
     set_program_var("environ", env);
     set_program_var("__elf_aux_vector", aux);
 
     /* Make a list of init functions to call. */
     objlist_init(&initlist);
     initlist_add_objects(obj_list, preload_tail, &initlist);
 
     r_debug_state(NULL, &obj_main->linkmap); /* say hello to gdb! */
 
     map_stacks_exec(NULL);
 
     dbg("resolving ifuncs");
     if (resolve_objects_ifunc(obj_main,
       ld_bind_now != NULL && *ld_bind_now != '\0', SYMLOOK_EARLY,
       NULL) == -1)
 	die();
 
     if (!obj_main->crt_no_init) {
 	/*
 	 * Make sure we don't call the main program's init and fini
 	 * functions for binaries linked with old crt1 which calls
 	 * _init itself.
 	 */
 	obj_main->init = obj_main->fini = (Elf_Addr)NULL;
 	obj_main->preinit_array = obj_main->init_array =
 	    obj_main->fini_array = (Elf_Addr)NULL;
     }
 
     wlock_acquire(rtld_bind_lock, &lockstate);
     if (obj_main->crt_no_init)
 	preinit_main();
     objlist_call_init(&initlist, &lockstate);
     _r_debug_postinit(&obj_main->linkmap);
     objlist_clear(&initlist);
     dbg("loading filtees");
     for (obj = obj_list->next; obj != NULL; obj = obj->next) {
 	if (ld_loadfltr || obj->z_loadfltr)
 	    load_filtees(obj, 0, &lockstate);
     }
     lock_release(rtld_bind_lock, &lockstate);
 
     dbg("transferring control to program entry point = %p", obj_main->entry);
 
     /* Return the exit procedure and the program entry point. */
     *exit_proc = rtld_exit;
     *objp = obj_main;
     return (func_ptr_type) obj_main->entry;
 }
 
 void *
 rtld_resolve_ifunc(const Obj_Entry *obj, const Elf_Sym *def)
 {
 	void *ptr;
 	Elf_Addr target;
 
 	ptr = (void *)make_function_pointer(def, obj);
 	target = ((Elf_Addr (*)(void))ptr)();
 	return ((void *)target);
 }
 
 Elf_Addr
 _rtld_bind(Obj_Entry *obj, Elf_Size reloff)
 {
     const Elf_Rel *rel;
     const Elf_Sym *def;
     const Obj_Entry *defobj;
     Elf_Addr *where;
     Elf_Addr target;
     RtldLockState lockstate;
 
     rlock_acquire(rtld_bind_lock, &lockstate);
     if (sigsetjmp(lockstate.env, 0) != 0)
 	    lock_upgrade(rtld_bind_lock, &lockstate);
     if (obj->pltrel)
 	rel = (const Elf_Rel *) ((caddr_t) obj->pltrel + reloff);
     else
 	rel = (const Elf_Rel *) ((caddr_t) obj->pltrela + reloff);
 
     where = (Elf_Addr *) (obj->relocbase + rel->r_offset);
     def = find_symdef(ELF_R_SYM(rel->r_info), obj, &defobj, true, NULL,
 	&lockstate);
     if (def == NULL)
 	die();
     if (ELF_ST_TYPE(def->st_info) == STT_GNU_IFUNC)
 	target = (Elf_Addr)rtld_resolve_ifunc(defobj, def);
     else
 	target = (Elf_Addr)(defobj->relocbase + def->st_value);
 
     dbg("\"%s\" in \"%s\" ==> %p in \"%s\"",
       defobj->strtab + def->st_name, basename(obj->path),
       (void *)target, basename(defobj->path));
 
     /*
      * Write the new contents for the jmpslot. Note that depending on
      * architecture, the value which we need to return back to the
      * lazy binding trampoline may or may not be the target
      * address. The value returned from reloc_jmpslot() is the value
      * that the trampoline needs.
      */
     target = reloc_jmpslot(where, target, defobj, obj, rel);
     lock_release(rtld_bind_lock, &lockstate);
     return target;
 }
 
 /*
  * Error reporting function.  Use it like printf.  If formats the message
  * into a buffer, and sets things up so that the next call to dlerror()
  * will return the message.
  */
 void
 _rtld_error(const char *fmt, ...)
 {
     static char buf[512];
     va_list ap;
 
     va_start(ap, fmt);
     rtld_vsnprintf(buf, sizeof buf, fmt, ap);
     error_message = buf;
     va_end(ap);
 }
 
 /*
  * Return a dynamically-allocated copy of the current error message, if any.
  */
 static char *
 errmsg_save(void)
 {
     return error_message == NULL ? NULL : xstrdup(error_message);
 }
 
 /*
  * Restore the current error message from a copy which was previously saved
  * by errmsg_save().  The copy is freed.
  */
 static void
 errmsg_restore(char *saved_msg)
 {
     if (saved_msg == NULL)
 	error_message = NULL;
     else {
 	_rtld_error("%s", saved_msg);
 	free(saved_msg);
     }
 }
 
 static const char *
 basename(const char *name)
 {
     const char *p = strrchr(name, '/');
     return p != NULL ? p + 1 : name;
 }
 
 static struct utsname uts;
 
 static char *
 origin_subst_one(char *real, const char *kw, const char *subst,
     bool may_free)
 {
 	char *p, *p1, *res, *resp;
 	int subst_len, kw_len, subst_count, old_len, new_len;
 
 	kw_len = strlen(kw);
 
 	/*
 	 * First, count the number of the keyword occurences, to
 	 * preallocate the final string.
 	 */
 	for (p = real, subst_count = 0;; p = p1 + kw_len, subst_count++) {
 		p1 = strstr(p, kw);
 		if (p1 == NULL)
 			break;
 	}
 
 	/*
 	 * If the keyword is not found, just return.
 	 */
 	if (subst_count == 0)
 		return (may_free ? real : xstrdup(real));
 
 	/*
 	 * There is indeed something to substitute.  Calculate the
 	 * length of the resulting string, and allocate it.
 	 */
 	subst_len = strlen(subst);
 	old_len = strlen(real);
 	new_len = old_len + (subst_len - kw_len) * subst_count;
 	res = xmalloc(new_len + 1);
 
 	/*
 	 * Now, execute the substitution loop.
 	 */
 	for (p = real, resp = res, *resp = '\0';;) {
 		p1 = strstr(p, kw);
 		if (p1 != NULL) {
 			/* Copy the prefix before keyword. */
 			memcpy(resp, p, p1 - p);
 			resp += p1 - p;
 			/* Keyword replacement. */
 			memcpy(resp, subst, subst_len);
 			resp += subst_len;
 			*resp = '\0';
 			p = p1 + kw_len;
 		} else
 			break;
 	}
 
 	/* Copy to the end of string and finish. */
 	strcat(resp, p);
 	if (may_free)
 		free(real);
 	return (res);
 }
 
 static char *
 origin_subst(char *real, const char *origin_path)
 {
 	char *res1, *res2, *res3, *res4;
 
 	if (uts.sysname[0] == '\0') {
 		if (uname(&uts) != 0) {
 			_rtld_error("utsname failed: %d", errno);
 			return (NULL);
 		}
 	}
 	res1 = origin_subst_one(real, "$ORIGIN", origin_path, false);
 	res2 = origin_subst_one(res1, "$OSNAME", uts.sysname, true);
 	res3 = origin_subst_one(res2, "$OSREL", uts.release, true);
 	res4 = origin_subst_one(res3, "$PLATFORM", uts.machine, true);
 	return (res4);
 }
 
 static void
 die(void)
 {
     const char *msg = dlerror();
 
     if (msg == NULL)
 	msg = "Fatal error";
     rtld_fdputstr(STDERR_FILENO, msg);
     rtld_fdputchar(STDERR_FILENO, '\n');
     _exit(1);
 }
 
 /*
  * Process a shared object's DYNAMIC section, and save the important
  * information in its Obj_Entry structure.
  */
 static void
 digest_dynamic1(Obj_Entry *obj, int early, const Elf_Dyn **dyn_rpath,
     const Elf_Dyn **dyn_soname, const Elf_Dyn **dyn_runpath)
 {
     const Elf_Dyn *dynp;
     Needed_Entry **needed_tail = &obj->needed;
     Needed_Entry **needed_filtees_tail = &obj->needed_filtees;
     Needed_Entry **needed_aux_filtees_tail = &obj->needed_aux_filtees;
     const Elf_Hashelt *hashtab;
     const Elf32_Word *hashval;
     Elf32_Word bkt, nmaskwords;
     int bloom_size32;
-    bool nmw_power2;
     int plttype = DT_REL;
 
     *dyn_rpath = NULL;
     *dyn_soname = NULL;
     *dyn_runpath = NULL;
 
     obj->bind_now = false;
     for (dynp = obj->dynamic;  dynp->d_tag != DT_NULL;  dynp++) {
 	switch (dynp->d_tag) {
 
 	case DT_REL:
 	    obj->rel = (const Elf_Rel *) (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_RELSZ:
 	    obj->relsize = dynp->d_un.d_val;
 	    break;
 
 	case DT_RELENT:
 	    assert(dynp->d_un.d_val == sizeof(Elf_Rel));
 	    break;
 
 	case DT_JMPREL:
 	    obj->pltrel = (const Elf_Rel *)
 	      (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_PLTRELSZ:
 	    obj->pltrelsize = dynp->d_un.d_val;
 	    break;
 
 	case DT_RELA:
 	    obj->rela = (const Elf_Rela *) (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_RELASZ:
 	    obj->relasize = dynp->d_un.d_val;
 	    break;
 
 	case DT_RELAENT:
 	    assert(dynp->d_un.d_val == sizeof(Elf_Rela));
 	    break;
 
 	case DT_PLTREL:
 	    plttype = dynp->d_un.d_val;
 	    assert(dynp->d_un.d_val == DT_REL || plttype == DT_RELA);
 	    break;
 
 	case DT_SYMTAB:
 	    obj->symtab = (const Elf_Sym *)
 	      (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_SYMENT:
 	    assert(dynp->d_un.d_val == sizeof(Elf_Sym));
 	    break;
 
 	case DT_STRTAB:
 	    obj->strtab = (const char *) (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_STRSZ:
 	    obj->strsize = dynp->d_un.d_val;
 	    break;
 
 	case DT_VERNEED:
 	    obj->verneed = (const Elf_Verneed *) (obj->relocbase +
 		dynp->d_un.d_val);
 	    break;
 
 	case DT_VERNEEDNUM:
 	    obj->verneednum = dynp->d_un.d_val;
 	    break;
 
 	case DT_VERDEF:
 	    obj->verdef = (const Elf_Verdef *) (obj->relocbase +
 		dynp->d_un.d_val);
 	    break;
 
 	case DT_VERDEFNUM:
 	    obj->verdefnum = dynp->d_un.d_val;
 	    break;
 
 	case DT_VERSYM:
 	    obj->versyms = (const Elf_Versym *)(obj->relocbase +
 		dynp->d_un.d_val);
 	    break;
 
 	case DT_HASH:
 	    {
 		hashtab = (const Elf_Hashelt *)(obj->relocbase +
 		    dynp->d_un.d_ptr);
 		obj->nbuckets = hashtab[0];
 		obj->nchains = hashtab[1];
 		obj->buckets = hashtab + 2;
 		obj->chains = obj->buckets + obj->nbuckets;
 		obj->valid_hash_sysv = obj->nbuckets > 0 && obj->nchains > 0 &&
 		  obj->buckets != NULL;
 	    }
 	    break;
 
 	case DT_GNU_HASH:
 	    {
 		hashtab = (const Elf_Hashelt *)(obj->relocbase +
 		    dynp->d_un.d_ptr);
 		obj->nbuckets_gnu = hashtab[0];
 		obj->symndx_gnu = hashtab[1];
 		nmaskwords = hashtab[2];
 		bloom_size32 = (__ELF_WORD_SIZE / 32) * nmaskwords;
-		/* Number of bitmask words is required to be power of 2 */
-		nmw_power2 = ((nmaskwords & (nmaskwords - 1)) == 0);
 		obj->maskwords_bm_gnu = nmaskwords - 1;
 		obj->shift2_gnu = hashtab[3];
 		obj->bloom_gnu = (Elf_Addr *) (hashtab + 4);
 		obj->buckets_gnu = hashtab + 4 + bloom_size32;
 		obj->chain_zero_gnu = obj->buckets_gnu + obj->nbuckets_gnu -
 		  obj->symndx_gnu;
-		obj->valid_hash_gnu = nmw_power2 && obj->nbuckets_gnu > 0 &&
-		  obj->buckets_gnu != NULL;
+		/* Number of bitmask words is required to be power of 2 */
+		obj->valid_hash_gnu = powerof2(nmaskwords) &&
+		    obj->nbuckets_gnu > 0 && obj->buckets_gnu != NULL;
 	    }
 	    break;
 
 	case DT_NEEDED:
 	    if (!obj->rtld) {
 		Needed_Entry *nep = NEW(Needed_Entry);
 		nep->name = dynp->d_un.d_val;
 		nep->obj = NULL;
 		nep->next = NULL;
 
 		*needed_tail = nep;
 		needed_tail = &nep->next;
 	    }
 	    break;
 
 	case DT_FILTER:
 	    if (!obj->rtld) {
 		Needed_Entry *nep = NEW(Needed_Entry);
 		nep->name = dynp->d_un.d_val;
 		nep->obj = NULL;
 		nep->next = NULL;
 
 		*needed_filtees_tail = nep;
 		needed_filtees_tail = &nep->next;
 	    }
 	    break;
 
 	case DT_AUXILIARY:
 	    if (!obj->rtld) {
 		Needed_Entry *nep = NEW(Needed_Entry);
 		nep->name = dynp->d_un.d_val;
 		nep->obj = NULL;
 		nep->next = NULL;
 
 		*needed_aux_filtees_tail = nep;
 		needed_aux_filtees_tail = &nep->next;
 	    }
 	    break;
 
 	case DT_PLTGOT:
 	    obj->pltgot = (Elf_Addr *) (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_TEXTREL:
 	    obj->textrel = true;
 	    break;
 
 	case DT_SYMBOLIC:
 	    obj->symbolic = true;
 	    break;
 
 	case DT_RPATH:
 	    /*
 	     * We have to wait until later to process this, because we
 	     * might not have gotten the address of the string table yet.
 	     */
 	    *dyn_rpath = dynp;
 	    break;
 
 	case DT_SONAME:
 	    *dyn_soname = dynp;
 	    break;
 
 	case DT_RUNPATH:
 	    *dyn_runpath = dynp;
 	    break;
 
 	case DT_INIT:
 	    obj->init = (Elf_Addr) (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_PREINIT_ARRAY:
 	    obj->preinit_array = (Elf_Addr)(obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_PREINIT_ARRAYSZ:
 	    obj->preinit_array_num = dynp->d_un.d_val / sizeof(Elf_Addr);
 	    break;
 
 	case DT_INIT_ARRAY:
 	    obj->init_array = (Elf_Addr)(obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_INIT_ARRAYSZ:
 	    obj->init_array_num = dynp->d_un.d_val / sizeof(Elf_Addr);
 	    break;
 
 	case DT_FINI:
 	    obj->fini = (Elf_Addr) (obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_FINI_ARRAY:
 	    obj->fini_array = (Elf_Addr)(obj->relocbase + dynp->d_un.d_ptr);
 	    break;
 
 	case DT_FINI_ARRAYSZ:
 	    obj->fini_array_num = dynp->d_un.d_val / sizeof(Elf_Addr);
 	    break;
 
 	/*
 	 * Don't process DT_DEBUG on MIPS as the dynamic section
 	 * is mapped read-only. DT_MIPS_RLD_MAP is used instead.
 	 */
 
 #ifndef __mips__
 	case DT_DEBUG:
 	    /* XXX - not implemented yet */
 	    if (!early)
 		dbg("Filling in DT_DEBUG entry");
 	    ((Elf_Dyn*)dynp)->d_un.d_ptr = (Elf_Addr) &r_debug;
 	    break;
 #endif
 
 	case DT_FLAGS:
 		if ((dynp->d_un.d_val & DF_ORIGIN) && trust)
 		    obj->z_origin = true;
 		if (dynp->d_un.d_val & DF_SYMBOLIC)
 		    obj->symbolic = true;
 		if (dynp->d_un.d_val & DF_TEXTREL)
 		    obj->textrel = true;
 		if (dynp->d_un.d_val & DF_BIND_NOW)
 		    obj->bind_now = true;
 		/*if (dynp->d_un.d_val & DF_STATIC_TLS)
 		    ;*/
 	    break;
 #ifdef __mips__
 	case DT_MIPS_LOCAL_GOTNO:
 		obj->local_gotno = dynp->d_un.d_val;
 	    break;
 
 	case DT_MIPS_SYMTABNO:
 		obj->symtabno = dynp->d_un.d_val;
 		break;
 
 	case DT_MIPS_GOTSYM:
 		obj->gotsym = dynp->d_un.d_val;
 		break;
 
 	case DT_MIPS_RLD_MAP:
 		*((Elf_Addr *)(dynp->d_un.d_ptr)) = (Elf_Addr) &r_debug;
 		break;
 #endif
 
 	case DT_FLAGS_1:
 		if (dynp->d_un.d_val & DF_1_NOOPEN)
 		    obj->z_noopen = true;
 		if ((dynp->d_un.d_val & DF_1_ORIGIN) && trust)
 		    obj->z_origin = true;
 		/*if (dynp->d_un.d_val & DF_1_GLOBAL)
 		    XXX ;*/
 		if (dynp->d_un.d_val & DF_1_BIND_NOW)
 		    obj->bind_now = true;
 		if (dynp->d_un.d_val & DF_1_NODELETE)
 		    obj->z_nodelete = true;
 		if (dynp->d_un.d_val & DF_1_LOADFLTR)
 		    obj->z_loadfltr = true;
 		if (dynp->d_un.d_val & DF_1_INTERPOSE)
 		    obj->z_interpose = true;
 		if (dynp->d_un.d_val & DF_1_NODEFLIB)
 		    obj->z_nodeflib = true;
 	    break;
 
 	default:
 	    if (!early) {
 		dbg("Ignoring d_tag %ld = %#lx", (long)dynp->d_tag,
 		    (long)dynp->d_tag);
 	    }
 	    break;
 	}
     }
 
     obj->traced = false;
 
     if (plttype == DT_RELA) {
 	obj->pltrela = (const Elf_Rela *) obj->pltrel;
 	obj->pltrel = NULL;
 	obj->pltrelasize = obj->pltrelsize;
 	obj->pltrelsize = 0;
     }
 
     /* Determine size of dynsym table (equal to nchains of sysv hash) */
     if (obj->valid_hash_sysv)
 	obj->dynsymcount = obj->nchains;
     else if (obj->valid_hash_gnu) {
 	obj->dynsymcount = 0;
 	for (bkt = 0; bkt < obj->nbuckets_gnu; bkt++) {
 	    if (obj->buckets_gnu[bkt] == 0)
 		continue;
 	    hashval = &obj->chain_zero_gnu[obj->buckets_gnu[bkt]];
 	    do
 		obj->dynsymcount++;
 	    while ((*hashval++ & 1u) == 0);
 	}
 	obj->dynsymcount += obj->symndx_gnu;
     }
 }
 
 static void
 digest_dynamic2(Obj_Entry *obj, const Elf_Dyn *dyn_rpath,
     const Elf_Dyn *dyn_soname, const Elf_Dyn *dyn_runpath)
 {
 
     if (obj->z_origin && obj->origin_path == NULL) {
 	obj->origin_path = xmalloc(PATH_MAX);
 	if (rtld_dirname_abs(obj->path, obj->origin_path) == -1)
 	    die();
     }
 
     if (dyn_runpath != NULL) {
 	obj->runpath = (char *)obj->strtab + dyn_runpath->d_un.d_val;
 	if (obj->z_origin)
 	    obj->runpath = origin_subst(obj->runpath, obj->origin_path);
     }
     else if (dyn_rpath != NULL) {
 	obj->rpath = (char *)obj->strtab + dyn_rpath->d_un.d_val;
 	if (obj->z_origin)
 	    obj->rpath = origin_subst(obj->rpath, obj->origin_path);
     }
 
     if (dyn_soname != NULL)
 	object_add_name(obj, obj->strtab + dyn_soname->d_un.d_val);
 }
 
 static void
 digest_dynamic(Obj_Entry *obj, int early)
 {
 	const Elf_Dyn *dyn_rpath;
 	const Elf_Dyn *dyn_soname;
 	const Elf_Dyn *dyn_runpath;
 
 	digest_dynamic1(obj, early, &dyn_rpath, &dyn_soname, &dyn_runpath);
 	digest_dynamic2(obj, dyn_rpath, dyn_soname, dyn_runpath);
 }
 
 /*
  * Process a shared object's program header.  This is used only for the
  * main program, when the kernel has already loaded the main program
  * into memory before calling the dynamic linker.  It creates and
  * returns an Obj_Entry structure.
  */
 static Obj_Entry *
 digest_phdr(const Elf_Phdr *phdr, int phnum, caddr_t entry, const char *path)
 {
     Obj_Entry *obj;
     const Elf_Phdr *phlimit = phdr + phnum;
     const Elf_Phdr *ph;
     Elf_Addr note_start, note_end;
     int nsegs = 0;
 
     obj = obj_new();
     for (ph = phdr;  ph < phlimit;  ph++) {
 	if (ph->p_type != PT_PHDR)
 	    continue;
 
 	obj->phdr = phdr;
 	obj->phsize = ph->p_memsz;
 	obj->relocbase = (caddr_t)phdr - ph->p_vaddr;
 	break;
     }
 
     obj->stack_flags = PF_X | PF_R | PF_W;
 
     for (ph = phdr;  ph < phlimit;  ph++) {
 	switch (ph->p_type) {
 
 	case PT_INTERP:
 	    obj->interp = (const char *)(ph->p_vaddr + obj->relocbase);
 	    break;
 
 	case PT_LOAD:
 	    if (nsegs == 0) {	/* First load segment */
 		obj->vaddrbase = trunc_page(ph->p_vaddr);
 		obj->mapbase = obj->vaddrbase + obj->relocbase;
 		obj->textsize = round_page(ph->p_vaddr + ph->p_memsz) -
 		  obj->vaddrbase;
 	    } else {		/* Last load segment */
 		obj->mapsize = round_page(ph->p_vaddr + ph->p_memsz) -
 		  obj->vaddrbase;
 	    }
 	    nsegs++;
 	    break;
 
 	case PT_DYNAMIC:
 	    obj->dynamic = (const Elf_Dyn *)(ph->p_vaddr + obj->relocbase);
 	    break;
 
 	case PT_TLS:
 	    obj->tlsindex = 1;
 	    obj->tlssize = ph->p_memsz;
 	    obj->tlsalign = ph->p_align;
 	    obj->tlsinitsize = ph->p_filesz;
 	    obj->tlsinit = (void*)(ph->p_vaddr + obj->relocbase);
 	    break;
 
 	case PT_GNU_STACK:
 	    obj->stack_flags = ph->p_flags;
 	    break;
 
 	case PT_GNU_RELRO:
 	    obj->relro_page = obj->relocbase + trunc_page(ph->p_vaddr);
 	    obj->relro_size = round_page(ph->p_memsz);
 	    break;
 
 	case PT_NOTE:
 	    note_start = (Elf_Addr)obj->relocbase + ph->p_vaddr;
 	    note_end = note_start + ph->p_filesz;
 	    digest_notes(obj, note_start, note_end);
 	    break;
 	}
     }
     if (nsegs < 1) {
 	_rtld_error("%s: too few PT_LOAD segments", path);
 	return NULL;
     }
 
     obj->entry = entry;
     return obj;
 }
 
 void
 digest_notes(Obj_Entry *obj, Elf_Addr note_start, Elf_Addr note_end)
 {
 	const Elf_Note *note;
 	const char *note_name;
 	uintptr_t p;
 
 	for (note = (const Elf_Note *)note_start; (Elf_Addr)note < note_end;
 	    note = (const Elf_Note *)((const char *)(note + 1) +
 	      roundup2(note->n_namesz, sizeof(Elf32_Addr)) +
 	      roundup2(note->n_descsz, sizeof(Elf32_Addr)))) {
 		if (note->n_namesz != sizeof(NOTE_FREEBSD_VENDOR) ||
 		    note->n_descsz != sizeof(int32_t))
 			continue;
 		if (note->n_type != ABI_NOTETYPE &&
 		    note->n_type != CRT_NOINIT_NOTETYPE)
 			continue;
 		note_name = (const char *)(note + 1);
 		if (strncmp(NOTE_FREEBSD_VENDOR, note_name,
 		    sizeof(NOTE_FREEBSD_VENDOR)) != 0)
 			continue;
 		switch (note->n_type) {
 		case ABI_NOTETYPE:
 			/* FreeBSD osrel note */
 			p = (uintptr_t)(note + 1);
 			p += roundup2(note->n_namesz, sizeof(Elf32_Addr));
 			obj->osrel = *(const int32_t *)(p);
 			dbg("note osrel %d", obj->osrel);
 			break;
 		case CRT_NOINIT_NOTETYPE:
 			/* FreeBSD 'crt does not call init' note */
 			obj->crt_no_init = true;
 			dbg("note crt_no_init");
 			break;
 		}
 	}
 }
 
 static Obj_Entry *
 dlcheck(void *handle)
 {
     Obj_Entry *obj;
 
     for (obj = obj_list;  obj != NULL;  obj = obj->next)
 	if (obj == (Obj_Entry *) handle)
 	    break;
 
     if (obj == NULL || obj->refcount == 0 || obj->dl_refcount == 0) {
 	_rtld_error("Invalid shared object handle %p", handle);
 	return NULL;
     }
     return obj;
 }
 
 /*
  * If the given object is already in the donelist, return true.  Otherwise
  * add the object to the list and return false.
  */
 static bool
 donelist_check(DoneList *dlp, const Obj_Entry *obj)
 {
     unsigned int i;
 
     for (i = 0;  i < dlp->num_used;  i++)
 	if (dlp->objs[i] == obj)
 	    return true;
     /*
      * Our donelist allocation should always be sufficient.  But if
      * our threads locking isn't working properly, more shared objects
      * could have been loaded since we allocated the list.  That should
      * never happen, but we'll handle it properly just in case it does.
      */
     if (dlp->num_used < dlp->num_alloc)
 	dlp->objs[dlp->num_used++] = obj;
     return false;
 }
 
 /*
  * Hash function for symbol table lookup.  Don't even think about changing
  * this.  It is specified by the System V ABI.
  */
 unsigned long
 elf_hash(const char *name)
 {
     const unsigned char *p = (const unsigned char *) name;
     unsigned long h = 0;
     unsigned long g;
 
     while (*p != '\0') {
 	h = (h << 4) + *p++;
 	if ((g = h & 0xf0000000) != 0)
 	    h ^= g >> 24;
 	h &= ~g;
     }
     return h;
 }
 
 /*
  * The GNU hash function is the Daniel J. Bernstein hash clipped to 32 bits
  * unsigned in case it's implemented with a wider type.
  */
 static uint32_t
 gnu_hash(const char *s)
 {
 	uint32_t h;
 	unsigned char c;
 
 	h = 5381;
 	for (c = *s; c != '\0'; c = *++s)
 		h = h * 33 + c;
 	return (h & 0xffffffff);
 }
 
 
 /*
  * Find the library with the given name, and return its full pathname.
  * The returned string is dynamically allocated.  Generates an error
  * message and returns NULL if the library cannot be found.
  *
  * If the second argument is non-NULL, then it refers to an already-
  * loaded shared object, whose library search path will be searched.
  *
  * If a library is successfully located via LD_LIBRARY_PATH_FDS, its
  * descriptor (which is close-on-exec) will be passed out via the third
  * argument.
  *
  * The search order is:
  *   DT_RPATH in the referencing file _unless_ DT_RUNPATH is present (1)
  *   DT_RPATH of the main object if DSO without defined DT_RUNPATH (1)
  *   LD_LIBRARY_PATH
  *   DT_RUNPATH in the referencing file
  *   ldconfig hints (if -z nodefaultlib, filter out default library directories
  *	 from list)
  *   /lib:/usr/lib _unless_ the referencing file is linked with -z nodefaultlib
  *
  * (1) Handled in digest_dynamic2 - rpath left NULL if runpath defined.
  */
 static char *
 find_library(const char *xname, const Obj_Entry *refobj, int *fdp)
 {
     char *pathname;
     char *name;
     bool nodeflib, objgiven;
 
     objgiven = refobj != NULL;
     if (strchr(xname, '/') != NULL) {	/* Hard coded pathname */
 	if (xname[0] != '/' && !trust) {
 	    _rtld_error("Absolute pathname required for shared object \"%s\"",
 	      xname);
 	    return NULL;
 	}
 	if (objgiven && refobj->z_origin) {
 		return (origin_subst(__DECONST(char *, xname),
 		    refobj->origin_path));
 	} else {
 		return (xstrdup(xname));
 	}
     }
 
     if (libmap_disable || !objgiven ||
 	(name = lm_find(refobj->path, xname)) == NULL)
 	name = (char *)xname;
 
     dbg(" Searching for \"%s\"", name);
 
     /*
      * If refobj->rpath != NULL, then refobj->runpath is NULL.  Fall
      * back to pre-conforming behaviour if user requested so with
      * LD_LIBRARY_PATH_RPATH environment variable and ignore -z
      * nodeflib.
      */
     if (objgiven && refobj->rpath != NULL && ld_library_path_rpath) {
 	if ((pathname = search_library_path(name, ld_library_path)) != NULL ||
 	  (refobj != NULL &&
 	  (pathname = search_library_path(name, refobj->rpath)) != NULL) ||
 	  (pathname = search_library_pathfds(name, ld_library_dirs, fdp)) != NULL ||
           (pathname = search_library_path(name, gethints(false))) != NULL ||
 	  (pathname = search_library_path(name, STANDARD_LIBRARY_PATH)) != NULL)
 	    return (pathname);
     } else {
 	nodeflib = objgiven ? refobj->z_nodeflib : false;
 	if ((objgiven &&
 	  (pathname = search_library_path(name, refobj->rpath)) != NULL) ||
 	  (objgiven && refobj->runpath == NULL && refobj != obj_main &&
 	  (pathname = search_library_path(name, obj_main->rpath)) != NULL) ||
 	  (pathname = search_library_path(name, ld_library_path)) != NULL ||
 	  (objgiven &&
 	  (pathname = search_library_path(name, refobj->runpath)) != NULL) ||
 	  (pathname = search_library_pathfds(name, ld_library_dirs, fdp)) != NULL ||
 	  (pathname = search_library_path(name, gethints(nodeflib))) != NULL ||
 	  (objgiven && !nodeflib &&
 	  (pathname = search_library_path(name, STANDARD_LIBRARY_PATH)) != NULL))
 	    return (pathname);
     }
 
     if (objgiven && refobj->path != NULL) {
 	_rtld_error("Shared object \"%s\" not found, required by \"%s\"",
 	  name, basename(refobj->path));
     } else {
 	_rtld_error("Shared object \"%s\" not found", name);
     }
     return NULL;
 }
 
 /*
  * Given a symbol number in a referencing object, find the corresponding
  * definition of the symbol.  Returns a pointer to the symbol, or NULL if
  * no definition was found.  Returns a pointer to the Obj_Entry of the
  * defining object via the reference parameter DEFOBJ_OUT.
  */
 const Elf_Sym *
 find_symdef(unsigned long symnum, const Obj_Entry *refobj,
     const Obj_Entry **defobj_out, int flags, SymCache *cache,
     RtldLockState *lockstate)
 {
     const Elf_Sym *ref;
     const Elf_Sym *def;
     const Obj_Entry *defobj;
     SymLook req;
     const char *name;
     int res;
 
     /*
      * If we have already found this symbol, get the information from
      * the cache.
      */
     if (symnum >= refobj->dynsymcount)
 	return NULL;	/* Bad object */
     if (cache != NULL && cache[symnum].sym != NULL) {
 	*defobj_out = cache[symnum].obj;
 	return cache[symnum].sym;
     }
 
     ref = refobj->symtab + symnum;
     name = refobj->strtab + ref->st_name;
     def = NULL;
     defobj = NULL;
 
     /*
      * We don't have to do a full scale lookup if the symbol is local.
      * We know it will bind to the instance in this load module; to
      * which we already have a pointer (ie ref). By not doing a lookup,
      * we not only improve performance, but it also avoids unresolvable
      * symbols when local symbols are not in the hash table. This has
      * been seen with the ia64 toolchain.
      */
     if (ELF_ST_BIND(ref->st_info) != STB_LOCAL) {
 	if (ELF_ST_TYPE(ref->st_info) == STT_SECTION) {
 	    _rtld_error("%s: Bogus symbol table entry %lu", refobj->path,
 		symnum);
 	}
 	symlook_init(&req, name);
 	req.flags = flags;
 	req.ventry = fetch_ventry(refobj, symnum);
 	req.lockstate = lockstate;
 	res = symlook_default(&req, refobj);
 	if (res == 0) {
 	    def = req.sym_out;
 	    defobj = req.defobj_out;
 	}
     } else {
 	def = ref;
 	defobj = refobj;
     }
 
     /*
      * If we found no definition and the reference is weak, treat the
      * symbol as having the value zero.
      */
     if (def == NULL && ELF_ST_BIND(ref->st_info) == STB_WEAK) {
 	def = &sym_zero;
 	defobj = obj_main;
     }
 
     if (def != NULL) {
 	*defobj_out = defobj;
 	/* Record the information in the cache to avoid subsequent lookups. */
 	if (cache != NULL) {
 	    cache[symnum].sym = def;
 	    cache[symnum].obj = defobj;
 	}
     } else {
 	if (refobj != &obj_rtld)
 	    _rtld_error("%s: Undefined symbol \"%s\"", refobj->path, name);
     }
     return def;
 }
 
 /*
  * Return the search path from the ldconfig hints file, reading it if
  * necessary.  If nostdlib is true, then the default search paths are
  * not added to result.
  *
  * Returns NULL if there are problems with the hints file,
  * or if the search path there is empty.
  */
 static const char *
 gethints(bool nostdlib)
 {
 	static char *hints, *filtered_path;
 	struct elfhints_hdr hdr;
 	struct fill_search_info_args sargs, hargs;
 	struct dl_serinfo smeta, hmeta, *SLPinfo, *hintinfo;
 	struct dl_serpath *SLPpath, *hintpath;
 	char *p;
 	unsigned int SLPndx, hintndx, fndx, fcount;
 	int fd;
 	size_t flen;
 	bool skip;
 
 	/* First call, read the hints file */
 	if (hints == NULL) {
 		/* Keep from trying again in case the hints file is bad. */
 		hints = "";
 
 		if ((fd = open(ld_elf_hints_path, O_RDONLY | O_CLOEXEC)) == -1)
 			return (NULL);
 		if (read(fd, &hdr, sizeof hdr) != sizeof hdr ||
 		    hdr.magic != ELFHINTS_MAGIC ||
 		    hdr.version != 1) {
 			close(fd);
 			return (NULL);
 		}
 		p = xmalloc(hdr.dirlistlen + 1);
 		if (lseek(fd, hdr.strtab + hdr.dirlist, SEEK_SET) == -1 ||
 		    read(fd, p, hdr.dirlistlen + 1) !=
 		    (ssize_t)hdr.dirlistlen + 1) {
 			free(p);
 			close(fd);
 			return (NULL);
 		}
 		hints = p;
 		close(fd);
 	}
 
 	/*
 	 * If caller agreed to receive list which includes the default
 	 * paths, we are done. Otherwise, if we still did not
 	 * calculated filtered result, do it now.
 	 */
 	if (!nostdlib)
 		return (hints[0] != '\0' ? hints : NULL);
 	if (filtered_path != NULL)
 		goto filt_ret;
 
 	/*
 	 * Obtain the list of all configured search paths, and the
 	 * list of the default paths.
 	 *
 	 * First estimate the size of the results.
 	 */
 	smeta.dls_size = __offsetof(struct dl_serinfo, dls_serpath);
 	smeta.dls_cnt = 0;
 	hmeta.dls_size = __offsetof(struct dl_serinfo, dls_serpath);
 	hmeta.dls_cnt = 0;
 
 	sargs.request = RTLD_DI_SERINFOSIZE;
 	sargs.serinfo = &smeta;
 	hargs.request = RTLD_DI_SERINFOSIZE;
 	hargs.serinfo = &hmeta;
 
 	path_enumerate(STANDARD_LIBRARY_PATH, fill_search_info, &sargs);
 	path_enumerate(p, fill_search_info, &hargs);
 
 	SLPinfo = xmalloc(smeta.dls_size);
 	hintinfo = xmalloc(hmeta.dls_size);
 
 	/*
 	 * Next fetch both sets of paths.
 	 */
 	sargs.request = RTLD_DI_SERINFO;
 	sargs.serinfo = SLPinfo;
 	sargs.serpath = &SLPinfo->dls_serpath[0];
 	sargs.strspace = (char *)&SLPinfo->dls_serpath[smeta.dls_cnt];
 
 	hargs.request = RTLD_DI_SERINFO;
 	hargs.serinfo = hintinfo;
 	hargs.serpath = &hintinfo->dls_serpath[0];
 	hargs.strspace = (char *)&hintinfo->dls_serpath[hmeta.dls_cnt];
 
 	path_enumerate(STANDARD_LIBRARY_PATH, fill_search_info, &sargs);
 	path_enumerate(p, fill_search_info, &hargs);
 
 	/*
 	 * Now calculate the difference between two sets, by excluding
 	 * standard paths from the full set.
 	 */
 	fndx = 0;
 	fcount = 0;
 	filtered_path = xmalloc(hdr.dirlistlen + 1);
 	hintpath = &hintinfo->dls_serpath[0];
 	for (hintndx = 0; hintndx < hmeta.dls_cnt; hintndx++, hintpath++) {
 		skip = false;
 		SLPpath = &SLPinfo->dls_serpath[0];
 		/*
 		 * Check each standard path against current.
 		 */
 		for (SLPndx = 0; SLPndx < smeta.dls_cnt; SLPndx++, SLPpath++) {
 			/* matched, skip the path */
 			if (!strcmp(hintpath->dls_name, SLPpath->dls_name)) {
 				skip = true;
 				break;
 			}
 		}
 		if (skip)
 			continue;
 		/*
 		 * Not matched against any standard path, add the path
 		 * to result. Separate consequtive paths with ':'.
 		 */
 		if (fcount > 0) {
 			filtered_path[fndx] = ':';
 			fndx++;
 		}
 		fcount++;
 		flen = strlen(hintpath->dls_name);
 		strncpy((filtered_path + fndx),	hintpath->dls_name, flen);
 		fndx += flen;
 	}
 	filtered_path[fndx] = '\0';
 
 	free(SLPinfo);
 	free(hintinfo);
 
 filt_ret:
 	return (filtered_path[0] != '\0' ? filtered_path : NULL);
 }
 
 static void
 init_dag(Obj_Entry *root)
 {
     const Needed_Entry *needed;
     const Objlist_Entry *elm;
     DoneList donelist;
 
     if (root->dag_inited)
 	return;
     donelist_init(&donelist);
 
     /* Root object belongs to own DAG. */
     objlist_push_tail(&root->dldags, root);
     objlist_push_tail(&root->dagmembers, root);
     donelist_check(&donelist, root);
 
     /*
      * Add dependencies of root object to DAG in breadth order
      * by exploiting the fact that each new object get added
      * to the tail of the dagmembers list.
      */
     STAILQ_FOREACH(elm, &root->dagmembers, link) {
 	for (needed = elm->obj->needed; needed != NULL; needed = needed->next) {
 	    if (needed->obj == NULL || donelist_check(&donelist, needed->obj))
 		continue;
 	    objlist_push_tail(&needed->obj->dldags, root);
 	    objlist_push_tail(&root->dagmembers, needed->obj);
 	}
     }
     root->dag_inited = true;
 }
 
 static void
 process_nodelete(Obj_Entry *root)
 {
 	const Objlist_Entry *elm;
 
 	/*
 	 * Walk over object DAG and process every dependent object that
 	 * is marked as DF_1_NODELETE. They need to grow their own DAG,
 	 * which then should have its reference upped separately.
 	 */
 	STAILQ_FOREACH(elm, &root->dagmembers, link) {
 		if (elm->obj != NULL && elm->obj->z_nodelete &&
 		    !elm->obj->ref_nodel) {
 			dbg("obj %s nodelete", elm->obj->path);
 			init_dag(elm->obj);
 			ref_dag(elm->obj);
 			elm->obj->ref_nodel = true;
 		}
 	}
 }
 /*
  * Initialize the dynamic linker.  The argument is the address at which
  * the dynamic linker has been mapped into memory.  The primary task of
  * this function is to relocate the dynamic linker.
  */
 static void
 init_rtld(caddr_t mapbase, Elf_Auxinfo **aux_info)
 {
     Obj_Entry objtmp;	/* Temporary rtld object */
     const Elf_Dyn *dyn_rpath;
     const Elf_Dyn *dyn_soname;
     const Elf_Dyn *dyn_runpath;
 
 #ifdef RTLD_INIT_PAGESIZES_EARLY
     /* The page size is required by the dynamic memory allocator. */
     init_pagesizes(aux_info);
 #endif
 
     /*
      * Conjure up an Obj_Entry structure for the dynamic linker.
      *
      * The "path" member can't be initialized yet because string constants
      * cannot yet be accessed. Below we will set it correctly.
      */
     memset(&objtmp, 0, sizeof(objtmp));
     objtmp.path = NULL;
     objtmp.rtld = true;
     objtmp.mapbase = mapbase;
 #ifdef PIC
     objtmp.relocbase = mapbase;
 #endif
     if (RTLD_IS_DYNAMIC()) {
 	objtmp.dynamic = rtld_dynamic(&objtmp);
 	digest_dynamic1(&objtmp, 1, &dyn_rpath, &dyn_soname, &dyn_runpath);
 	assert(objtmp.needed == NULL);
 #if !defined(__mips__)
 	/* MIPS has a bogus DT_TEXTREL. */
 	assert(!objtmp.textrel);
 #endif
 
 	/*
 	 * Temporarily put the dynamic linker entry into the object list, so
 	 * that symbols can be found.
 	 */
 
 	relocate_objects(&objtmp, true, &objtmp, 0, NULL);
     }
 
     /* Initialize the object list. */
     obj_tail = &obj_list;
 
     /* Now that non-local variables can be accesses, copy out obj_rtld. */
     memcpy(&obj_rtld, &objtmp, sizeof(obj_rtld));
 
 #ifndef RTLD_INIT_PAGESIZES_EARLY
     /* The page size is required by the dynamic memory allocator. */
     init_pagesizes(aux_info);
 #endif
 
     if (aux_info[AT_OSRELDATE] != NULL)
 	    osreldate = aux_info[AT_OSRELDATE]->a_un.a_val;
 
     digest_dynamic2(&obj_rtld, dyn_rpath, dyn_soname, dyn_runpath);
 
     /* Replace the path with a dynamically allocated copy. */
     obj_rtld.path = xstrdup(PATH_RTLD);
 
     r_debug.r_brk = r_debug_state;
     r_debug.r_state = RT_CONSISTENT;
 }
 
 /*
  * Retrieve the array of supported page sizes.  The kernel provides the page
  * sizes in increasing order.
  */
 static void
 init_pagesizes(Elf_Auxinfo **aux_info)
 {
 	static size_t psa[MAXPAGESIZES];
 	int mib[2];
 	size_t len, size;
 
 	if (aux_info[AT_PAGESIZES] != NULL && aux_info[AT_PAGESIZESLEN] !=
 	    NULL) {
 		size = aux_info[AT_PAGESIZESLEN]->a_un.a_val;
 		pagesizes = aux_info[AT_PAGESIZES]->a_un.a_ptr;
 	} else {
 		len = 2;
 		if (sysctlnametomib("hw.pagesizes", mib, &len) == 0)
 			size = sizeof(psa);
 		else {
 			/* As a fallback, retrieve the base page size. */
 			size = sizeof(psa[0]);
 			if (aux_info[AT_PAGESZ] != NULL) {
 				psa[0] = aux_info[AT_PAGESZ]->a_un.a_val;
 				goto psa_filled;
 			} else {
 				mib[0] = CTL_HW;
 				mib[1] = HW_PAGESIZE;
 				len = 2;
 			}
 		}
 		if (sysctl(mib, len, psa, &size, NULL, 0) == -1) {
 			_rtld_error("sysctl for hw.pagesize(s) failed");
 			die();
 		}
 psa_filled:
 		pagesizes = psa;
 	}
 	npagesizes = size / sizeof(pagesizes[0]);
 	/* Discard any invalid entries at the end of the array. */
 	while (npagesizes > 0 && pagesizes[npagesizes - 1] == 0)
 		npagesizes--;
 }
 
 /*
  * Add the init functions from a needed object list (and its recursive
  * needed objects) to "list".  This is not used directly; it is a helper
  * function for initlist_add_objects().  The write lock must be held
  * when this function is called.
  */
 static void
 initlist_add_neededs(Needed_Entry *needed, Objlist *list)
 {
     /* Recursively process the successor needed objects. */
     if (needed->next != NULL)
 	initlist_add_neededs(needed->next, list);
 
     /* Process the current needed object. */
     if (needed->obj != NULL)
 	initlist_add_objects(needed->obj, &needed->obj->next, list);
 }
 
 /*
  * Scan all of the DAGs rooted in the range of objects from "obj" to
  * "tail" and add their init functions to "list".  This recurses over
  * the DAGs and ensure the proper init ordering such that each object's
  * needed libraries are initialized before the object itself.  At the
  * same time, this function adds the objects to the global finalization
  * list "list_fini" in the opposite order.  The write lock must be
  * held when this function is called.
  */
 static void
 initlist_add_objects(Obj_Entry *obj, Obj_Entry **tail, Objlist *list)
 {
 
     if (obj->init_scanned || obj->init_done)
 	return;
     obj->init_scanned = true;
 
     /* Recursively process the successor objects. */
     if (&obj->next != tail)
 	initlist_add_objects(obj->next, tail, list);
 
     /* Recursively process the needed objects. */
     if (obj->needed != NULL)
 	initlist_add_neededs(obj->needed, list);
     if (obj->needed_filtees != NULL)
 	initlist_add_neededs(obj->needed_filtees, list);
     if (obj->needed_aux_filtees != NULL)
 	initlist_add_neededs(obj->needed_aux_filtees, list);
 
     /* Add the object to the init list. */
     if (obj->preinit_array != (Elf_Addr)NULL || obj->init != (Elf_Addr)NULL ||
       obj->init_array != (Elf_Addr)NULL)
 	objlist_push_tail(list, obj);
 
     /* Add the object to the global fini list in the reverse order. */
     if ((obj->fini != (Elf_Addr)NULL || obj->fini_array != (Elf_Addr)NULL)
       && !obj->on_fini_list) {
 	objlist_push_head(&list_fini, obj);
 	obj->on_fini_list = true;
     }
 }
 
 #ifndef FPTR_TARGET
 #define FPTR_TARGET(f)	((Elf_Addr) (f))
 #endif
 
 static void
 free_needed_filtees(Needed_Entry *n)
 {
     Needed_Entry *needed, *needed1;
 
     for (needed = n; needed != NULL; needed = needed->next) {
 	if (needed->obj != NULL) {
 	    dlclose(needed->obj);
 	    needed->obj = NULL;
 	}
     }
     for (needed = n; needed != NULL; needed = needed1) {
 	needed1 = needed->next;
 	free(needed);
     }
 }
 
 static void
 unload_filtees(Obj_Entry *obj)
 {
 
     free_needed_filtees(obj->needed_filtees);
     obj->needed_filtees = NULL;
     free_needed_filtees(obj->needed_aux_filtees);
     obj->needed_aux_filtees = NULL;
     obj->filtees_loaded = false;
 }
 
 static void
 load_filtee1(Obj_Entry *obj, Needed_Entry *needed, int flags,
     RtldLockState *lockstate)
 {
 
     for (; needed != NULL; needed = needed->next) {
 	needed->obj = dlopen_object(obj->strtab + needed->name, -1, obj,
 	  flags, ((ld_loadfltr || obj->z_loadfltr) ? RTLD_NOW : RTLD_LAZY) |
 	  RTLD_LOCAL, lockstate);
     }
 }
 
 static void
 load_filtees(Obj_Entry *obj, int flags, RtldLockState *lockstate)
 {
 
     lock_restart_for_upgrade(lockstate);
     if (!obj->filtees_loaded) {
 	load_filtee1(obj, obj->needed_filtees, flags, lockstate);
 	load_filtee1(obj, obj->needed_aux_filtees, flags, lockstate);
 	obj->filtees_loaded = true;
     }
 }
 
 static int
 process_needed(Obj_Entry *obj, Needed_Entry *needed, int flags)
 {
     Obj_Entry *obj1;
 
     for (; needed != NULL; needed = needed->next) {
 	obj1 = needed->obj = load_object(obj->strtab + needed->name, -1, obj,
 	  flags & ~RTLD_LO_NOLOAD);
 	if (obj1 == NULL && !ld_tracing && (flags & RTLD_LO_FILTEES) == 0)
 	    return (-1);
     }
     return (0);
 }
 
 /*
  * Given a shared object, traverse its list of needed objects, and load
  * each of them.  Returns 0 on success.  Generates an error message and
  * returns -1 on failure.
  */
 static int
 load_needed_objects(Obj_Entry *first, int flags)
 {
     Obj_Entry *obj;
 
     for (obj = first;  obj != NULL;  obj = obj->next) {
 	if (process_needed(obj, obj->needed, flags) == -1)
 	    return (-1);
     }
     return (0);
 }
 
 static int
 load_preload_objects(void)
 {
     char *p = ld_preload;
     Obj_Entry *obj;
     static const char delim[] = " \t:;";
 
     if (p == NULL)
 	return 0;
 
     p += strspn(p, delim);
     while (*p != '\0') {
 	size_t len = strcspn(p, delim);
 	char savech;
 
 	savech = p[len];
 	p[len] = '\0';
 	obj = load_object(p, -1, NULL, 0);
 	if (obj == NULL)
 	    return -1;	/* XXX - cleanup */
 	obj->z_interpose = true;
 	p[len] = savech;
 	p += len;
 	p += strspn(p, delim);
     }
     LD_UTRACE(UTRACE_PRELOAD_FINISHED, NULL, NULL, 0, 0, NULL);
     return 0;
 }
 
 static const char *
 printable_path(const char *path)
 {
 
 	return (path == NULL ? "<unknown>" : path);
 }
 
 /*
  * Load a shared object into memory, if it is not already loaded.  The
  * object may be specified by name or by user-supplied file descriptor
  * fd_u. In the later case, the fd_u descriptor is not closed, but its
  * duplicate is.
  *
  * Returns a pointer to the Obj_Entry for the object.  Returns NULL
  * on failure.
  */
 static Obj_Entry *
 load_object(const char *name, int fd_u, const Obj_Entry *refobj, int flags)
 {
     Obj_Entry *obj;
     int fd;
     struct stat sb;
     char *path;
 
     fd = -1;
     if (name != NULL) {
 	for (obj = obj_list->next;  obj != NULL;  obj = obj->next) {
 	    if (object_match_name(obj, name))
 		return (obj);
 	}
 
 	path = find_library(name, refobj, &fd);
 	if (path == NULL)
 	    return (NULL);
     } else
 	path = NULL;
 
     if (fd >= 0) {
 	/*
 	 * search_library_pathfds() opens a fresh file descriptor for the
 	 * library, so there is no need to dup().
 	 */
     } else if (fd_u == -1) {
 	/*
 	 * If we didn't find a match by pathname, or the name is not
 	 * supplied, open the file and check again by device and inode.
 	 * This avoids false mismatches caused by multiple links or ".."
 	 * in pathnames.
 	 *
 	 * To avoid a race, we open the file and use fstat() rather than
 	 * using stat().
 	 */
 	if ((fd = open(path, O_RDONLY | O_CLOEXEC)) == -1) {
 	    _rtld_error("Cannot open \"%s\"", path);
 	    free(path);
 	    return (NULL);
 	}
     } else {
 	fd = fcntl(fd_u, F_DUPFD_CLOEXEC, 0);
 	if (fd == -1) {
 	    _rtld_error("Cannot dup fd");
 	    free(path);
 	    return (NULL);
 	}
     }
     if (fstat(fd, &sb) == -1) {
 	_rtld_error("Cannot fstat \"%s\"", printable_path(path));
 	close(fd);
 	free(path);
 	return NULL;
     }
     for (obj = obj_list->next;  obj != NULL;  obj = obj->next)
 	if (obj->ino == sb.st_ino && obj->dev == sb.st_dev)
 	    break;
     if (obj != NULL && name != NULL) {
 	object_add_name(obj, name);
 	free(path);
 	close(fd);
 	return obj;
     }
     if (flags & RTLD_LO_NOLOAD) {
 	free(path);
 	close(fd);
 	return (NULL);
     }
 
     /* First use of this object, so we must map it in */
     obj = do_load_object(fd, name, path, &sb, flags);
     if (obj == NULL)
 	free(path);
     close(fd);
 
     return obj;
 }
 
 static Obj_Entry *
 do_load_object(int fd, const char *name, char *path, struct stat *sbp,
   int flags)
 {
     Obj_Entry *obj;
     struct statfs fs;
 
     /*
      * but first, make sure that environment variables haven't been
      * used to circumvent the noexec flag on a filesystem.
      */
     if (dangerous_ld_env) {
 	if (fstatfs(fd, &fs) != 0) {
 	    _rtld_error("Cannot fstatfs \"%s\"", printable_path(path));
 	    return NULL;
 	}
 	if (fs.f_flags & MNT_NOEXEC) {
 	    _rtld_error("Cannot execute objects on %s\n", fs.f_mntonname);
 	    return NULL;
 	}
     }
     dbg("loading \"%s\"", printable_path(path));
     obj = map_object(fd, printable_path(path), sbp);
     if (obj == NULL)
         return NULL;
 
     /*
      * If DT_SONAME is present in the object, digest_dynamic2 already
      * added it to the object names.
      */
     if (name != NULL)
 	object_add_name(obj, name);
     obj->path = path;
     digest_dynamic(obj, 0);
     dbg("%s valid_hash_sysv %d valid_hash_gnu %d dynsymcount %d", obj->path,
 	obj->valid_hash_sysv, obj->valid_hash_gnu, obj->dynsymcount);
     if (obj->z_noopen && (flags & (RTLD_LO_DLOPEN | RTLD_LO_TRACE)) ==
       RTLD_LO_DLOPEN) {
 	dbg("refusing to load non-loadable \"%s\"", obj->path);
 	_rtld_error("Cannot dlopen non-loadable %s", obj->path);
 	munmap(obj->mapbase, obj->mapsize);
 	obj_free(obj);
 	return (NULL);
     }
 
     obj->dlopened = (flags & RTLD_LO_DLOPEN) != 0;
     *obj_tail = obj;
     obj_tail = &obj->next;
     obj_count++;
     obj_loads++;
     linkmap_add(obj);	/* for GDB & dlinfo() */
     max_stack_flags |= obj->stack_flags;
 
     dbg("  %p .. %p: %s", obj->mapbase,
          obj->mapbase + obj->mapsize - 1, obj->path);
     if (obj->textrel)
 	dbg("  WARNING: %s has impure text", obj->path);
     LD_UTRACE(UTRACE_LOAD_OBJECT, obj, obj->mapbase, obj->mapsize, 0,
 	obj->path);    
 
     return obj;
 }
 
 static Obj_Entry *
 obj_from_addr(const void *addr)
 {
     Obj_Entry *obj;
 
     for (obj = obj_list;  obj != NULL;  obj = obj->next) {
 	if (addr < (void *) obj->mapbase)
 	    continue;
 	if (addr < (void *) (obj->mapbase + obj->mapsize))
 	    return obj;
     }
     return NULL;
 }
 
 static void
 preinit_main(void)
 {
     Elf_Addr *preinit_addr;
     int index;
 
     preinit_addr = (Elf_Addr *)obj_main->preinit_array;
     if (preinit_addr == NULL)
 	return;
 
     for (index = 0; index < obj_main->preinit_array_num; index++) {
 	if (preinit_addr[index] != 0 && preinit_addr[index] != 1) {
 	    dbg("calling preinit function for %s at %p", obj_main->path,
 	      (void *)preinit_addr[index]);
 	    LD_UTRACE(UTRACE_INIT_CALL, obj_main, (void *)preinit_addr[index],
 	      0, 0, obj_main->path);
 	    call_init_pointer(obj_main, preinit_addr[index]);
 	}
     }
 }
 
 /*
  * Call the finalization functions for each of the objects in "list"
  * belonging to the DAG of "root" and referenced once. If NULL "root"
  * is specified, every finalization function will be called regardless
  * of the reference count and the list elements won't be freed. All of
  * the objects are expected to have non-NULL fini functions.
  */
 static void
 objlist_call_fini(Objlist *list, Obj_Entry *root, RtldLockState *lockstate)
 {
     Objlist_Entry *elm;
     char *saved_msg;
     Elf_Addr *fini_addr;
     int index;
 
     assert(root == NULL || root->refcount == 1);
 
     /*
      * Preserve the current error message since a fini function might
      * call into the dynamic linker and overwrite it.
      */
     saved_msg = errmsg_save();
     do {
 	STAILQ_FOREACH(elm, list, link) {
 	    if (root != NULL && (elm->obj->refcount != 1 ||
 	      objlist_find(&root->dagmembers, elm->obj) == NULL))
 		continue;
 	    /* Remove object from fini list to prevent recursive invocation. */
 	    STAILQ_REMOVE(list, elm, Struct_Objlist_Entry, link);
 	    /*
 	     * XXX: If a dlopen() call references an object while the
 	     * fini function is in progress, we might end up trying to
 	     * unload the referenced object in dlclose() or the object
 	     * won't be unloaded although its fini function has been
 	     * called.
 	     */
 	    lock_release(rtld_bind_lock, lockstate);
 
 	    /*
 	     * It is legal to have both DT_FINI and DT_FINI_ARRAY defined.
 	     * When this happens, DT_FINI_ARRAY is processed first.
 	     */
 	    fini_addr = (Elf_Addr *)elm->obj->fini_array;
 	    if (fini_addr != NULL && elm->obj->fini_array_num > 0) {
 		for (index = elm->obj->fini_array_num - 1; index >= 0;
 		  index--) {
 		    if (fini_addr[index] != 0 && fini_addr[index] != 1) {
 			dbg("calling fini function for %s at %p",
 			    elm->obj->path, (void *)fini_addr[index]);
 			LD_UTRACE(UTRACE_FINI_CALL, elm->obj,
 			    (void *)fini_addr[index], 0, 0, elm->obj->path);
 			call_initfini_pointer(elm->obj, fini_addr[index]);
 		    }
 		}
 	    }
 	    if (elm->obj->fini != (Elf_Addr)NULL) {
 		dbg("calling fini function for %s at %p", elm->obj->path,
 		    (void *)elm->obj->fini);
 		LD_UTRACE(UTRACE_FINI_CALL, elm->obj, (void *)elm->obj->fini,
 		    0, 0, elm->obj->path);
 		call_initfini_pointer(elm->obj, elm->obj->fini);
 	    }
 	    wlock_acquire(rtld_bind_lock, lockstate);
 	    /* No need to free anything if process is going down. */
 	    if (root != NULL)
 	    	free(elm);
 	    /*
 	     * We must restart the list traversal after every fini call
 	     * because a dlclose() call from the fini function or from
 	     * another thread might have modified the reference counts.
 	     */
 	    break;
 	}
     } while (elm != NULL);
     errmsg_restore(saved_msg);
 }
 
 /*
  * Call the initialization functions for each of the objects in
  * "list".  All of the objects are expected to have non-NULL init
  * functions.
  */
 static void
 objlist_call_init(Objlist *list, RtldLockState *lockstate)
 {
     Objlist_Entry *elm;
     Obj_Entry *obj;
     char *saved_msg;
     Elf_Addr *init_addr;
     int index;
 
     /*
      * Clean init_scanned flag so that objects can be rechecked and
      * possibly initialized earlier if any of vectors called below
      * cause the change by using dlopen.
      */
     for (obj = obj_list;  obj != NULL;  obj = obj->next)
 	obj->init_scanned = false;
 
     /*
      * Preserve the current error message since an init function might
      * call into the dynamic linker and overwrite it.
      */
     saved_msg = errmsg_save();
     STAILQ_FOREACH(elm, list, link) {
 	if (elm->obj->init_done) /* Initialized early. */
 	    continue;
 	/*
 	 * Race: other thread might try to use this object before current
 	 * one completes the initilization. Not much can be done here
 	 * without better locking.
 	 */
 	elm->obj->init_done = true;
 	lock_release(rtld_bind_lock, lockstate);
 
         /*
          * It is legal to have both DT_INIT and DT_INIT_ARRAY defined.
          * When this happens, DT_INIT is processed first.
          */
 	if (elm->obj->init != (Elf_Addr)NULL) {
 	    dbg("calling init function for %s at %p", elm->obj->path,
 	        (void *)elm->obj->init);
 	    LD_UTRACE(UTRACE_INIT_CALL, elm->obj, (void *)elm->obj->init,
 	        0, 0, elm->obj->path);
 	    call_initfini_pointer(elm->obj, elm->obj->init);
 	}
 	init_addr = (Elf_Addr *)elm->obj->init_array;
 	if (init_addr != NULL) {
 	    for (index = 0; index < elm->obj->init_array_num; index++) {
 		if (init_addr[index] != 0 && init_addr[index] != 1) {
 		    dbg("calling init function for %s at %p", elm->obj->path,
 			(void *)init_addr[index]);
 		    LD_UTRACE(UTRACE_INIT_CALL, elm->obj,
 			(void *)init_addr[index], 0, 0, elm->obj->path);
 		    call_init_pointer(elm->obj, init_addr[index]);
 		}
 	    }
 	}
 	wlock_acquire(rtld_bind_lock, lockstate);
     }
     errmsg_restore(saved_msg);
 }
 
 static void
 objlist_clear(Objlist *list)
 {
     Objlist_Entry *elm;
 
     while (!STAILQ_EMPTY(list)) {
 	elm = STAILQ_FIRST(list);
 	STAILQ_REMOVE_HEAD(list, link);
 	free(elm);
     }
 }
 
 static Objlist_Entry *
 objlist_find(Objlist *list, const Obj_Entry *obj)
 {
     Objlist_Entry *elm;
 
     STAILQ_FOREACH(elm, list, link)
 	if (elm->obj == obj)
 	    return elm;
     return NULL;
 }
 
 static void
 objlist_init(Objlist *list)
 {
     STAILQ_INIT(list);
 }
 
 static void
 objlist_push_head(Objlist *list, Obj_Entry *obj)
 {
     Objlist_Entry *elm;
 
     elm = NEW(Objlist_Entry);
     elm->obj = obj;
     STAILQ_INSERT_HEAD(list, elm, link);
 }
 
 static void
 objlist_push_tail(Objlist *list, Obj_Entry *obj)
 {
     Objlist_Entry *elm;
 
     elm = NEW(Objlist_Entry);
     elm->obj = obj;
     STAILQ_INSERT_TAIL(list, elm, link);
 }
 
 static void
 objlist_put_after(Objlist *list, Obj_Entry *listobj, Obj_Entry *obj)
 {
 	Objlist_Entry *elm, *listelm;
 
 	STAILQ_FOREACH(listelm, list, link) {
 		if (listelm->obj == listobj)
 			break;
 	}
 	elm = NEW(Objlist_Entry);
 	elm->obj = obj;
 	if (listelm != NULL)
 		STAILQ_INSERT_AFTER(list, listelm, elm, link);
 	else
 		STAILQ_INSERT_TAIL(list, elm, link);
 }
 
 static void
 objlist_remove(Objlist *list, Obj_Entry *obj)
 {
     Objlist_Entry *elm;
 
     if ((elm = objlist_find(list, obj)) != NULL) {
 	STAILQ_REMOVE(list, elm, Struct_Objlist_Entry, link);
 	free(elm);
     }
 }
 
 /*
  * Relocate dag rooted in the specified object.
  * Returns 0 on success, or -1 on failure.
  */
 
 static int
 relocate_object_dag(Obj_Entry *root, bool bind_now, Obj_Entry *rtldobj,
     int flags, RtldLockState *lockstate)
 {
 	Objlist_Entry *elm;
 	int error;
 
 	error = 0;
 	STAILQ_FOREACH(elm, &root->dagmembers, link) {
 		error = relocate_object(elm->obj, bind_now, rtldobj, flags,
 		    lockstate);
 		if (error == -1)
 			break;
 	}
 	return (error);
 }
 
 /*
  * Relocate single object.
  * Returns 0 on success, or -1 on failure.
  */
 static int
 relocate_object(Obj_Entry *obj, bool bind_now, Obj_Entry *rtldobj,
     int flags, RtldLockState *lockstate)
 {
 
 	if (obj->relocated)
 		return (0);
 	obj->relocated = true;
 	if (obj != rtldobj)
 		dbg("relocating \"%s\"", obj->path);
 
 	if (obj->symtab == NULL || obj->strtab == NULL ||
 	    !(obj->valid_hash_sysv || obj->valid_hash_gnu)) {
 		_rtld_error("%s: Shared object has no run-time symbol table",
 			    obj->path);
 		return (-1);
 	}
 
 	if (obj->textrel) {
 		/* There are relocations to the write-protected text segment. */
 		if (mprotect(obj->mapbase, obj->textsize,
 		    PROT_READ|PROT_WRITE|PROT_EXEC) == -1) {
 			_rtld_error("%s: Cannot write-enable text segment: %s",
 			    obj->path, rtld_strerror(errno));
 			return (-1);
 		}
 	}
 
 	/* Process the non-PLT non-IFUNC relocations. */
 	if (reloc_non_plt(obj, rtldobj, flags, lockstate))
 		return (-1);
 
 	if (obj->textrel) {	/* Re-protected the text segment. */
 		if (mprotect(obj->mapbase, obj->textsize,
 		    PROT_READ|PROT_EXEC) == -1) {
 			_rtld_error("%s: Cannot write-protect text segment: %s",
 			    obj->path, rtld_strerror(errno));
 			return (-1);
 		}
 	}
 
 	/* Set the special PLT or GOT entries. */
 	init_pltgot(obj);
 
 	/* Process the PLT relocations. */
 	if (reloc_plt(obj) == -1)
 		return (-1);
 	/* Relocate the jump slots if we are doing immediate binding. */
 	if (obj->bind_now || bind_now)
 		if (reloc_jmpslots(obj, flags, lockstate) == -1)
 			return (-1);
 
 	/*
 	 * Process the non-PLT IFUNC relocations.  The relocations are
 	 * processed in two phases, because IFUNC resolvers may
 	 * reference other symbols, which must be readily processed
 	 * before resolvers are called.
 	 */
 	if (obj->non_plt_gnu_ifunc &&
 	    reloc_non_plt(obj, rtldobj, flags | SYMLOOK_IFUNC, lockstate))
 		return (-1);
 
 	if (obj->relro_size > 0) {
 		if (mprotect(obj->relro_page, obj->relro_size,
 		    PROT_READ) == -1) {
 			_rtld_error("%s: Cannot enforce relro protection: %s",
 			    obj->path, rtld_strerror(errno));
 			return (-1);
 		}
 	}
 
 	/*
 	 * Set up the magic number and version in the Obj_Entry.  These
 	 * were checked in the crt1.o from the original ElfKit, so we
 	 * set them for backward compatibility.
 	 */
 	obj->magic = RTLD_MAGIC;
 	obj->version = RTLD_VERSION;
 
 	return (0);
 }
 
 /*
  * Relocate newly-loaded shared objects.  The argument is a pointer to
  * the Obj_Entry for the first such object.  All objects from the first
  * to the end of the list of objects are relocated.  Returns 0 on success,
  * or -1 on failure.
  */
 static int
 relocate_objects(Obj_Entry *first, bool bind_now, Obj_Entry *rtldobj,
     int flags, RtldLockState *lockstate)
 {
 	Obj_Entry *obj;
 	int error;
 
 	for (error = 0, obj = first;  obj != NULL;  obj = obj->next) {
 		error = relocate_object(obj, bind_now, rtldobj, flags,
 		    lockstate);
 		if (error == -1)
 			break;
 	}
 	return (error);
 }
 
 /*
  * The handling of R_MACHINE_IRELATIVE relocations and jumpslots
  * referencing STT_GNU_IFUNC symbols is postponed till the other
  * relocations are done.  The indirect functions specified as
  * ifunc are allowed to call other symbols, so we need to have
  * objects relocated before asking for resolution from indirects.
  *
  * The R_MACHINE_IRELATIVE slots are resolved in greedy fashion,
  * instead of the usual lazy handling of PLT slots.  It is
  * consistent with how GNU does it.
  */
 static int
 resolve_object_ifunc(Obj_Entry *obj, bool bind_now, int flags,
     RtldLockState *lockstate)
 {
 	if (obj->irelative && reloc_iresolve(obj, lockstate) == -1)
 		return (-1);
 	if ((obj->bind_now || bind_now) && obj->gnu_ifunc &&
 	    reloc_gnu_ifunc(obj, flags, lockstate) == -1)
 		return (-1);
 	return (0);
 }
 
 static int
 resolve_objects_ifunc(Obj_Entry *first, bool bind_now, int flags,
     RtldLockState *lockstate)
 {
 	Obj_Entry *obj;
 
 	for (obj = first;  obj != NULL;  obj = obj->next) {
 		if (resolve_object_ifunc(obj, bind_now, flags, lockstate) == -1)
 			return (-1);
 	}
 	return (0);
 }
 
 static int
 initlist_objects_ifunc(Objlist *list, bool bind_now, int flags,
     RtldLockState *lockstate)
 {
 	Objlist_Entry *elm;
 
 	STAILQ_FOREACH(elm, list, link) {
 		if (resolve_object_ifunc(elm->obj, bind_now, flags,
 		    lockstate) == -1)
 			return (-1);
 	}
 	return (0);
 }
 
 /*
  * Cleanup procedure.  It will be called (by the atexit mechanism) just
  * before the process exits.
  */
 static void
 rtld_exit(void)
 {
     RtldLockState lockstate;
 
     wlock_acquire(rtld_bind_lock, &lockstate);
     dbg("rtld_exit()");
     objlist_call_fini(&list_fini, NULL, &lockstate);
     /* No need to remove the items from the list, since we are exiting. */
     if (!libmap_disable)
         lm_fini();
     lock_release(rtld_bind_lock, &lockstate);
 }
 
 /*
  * Iterate over a search path, translate each element, and invoke the
  * callback on the result.
  */
 static void *
 path_enumerate(const char *path, path_enum_proc callback, void *arg)
 {
     const char *trans;
     if (path == NULL)
 	return (NULL);
 
     path += strspn(path, ":;");
     while (*path != '\0') {
 	size_t len;
 	char  *res;
 
 	len = strcspn(path, ":;");
 	trans = lm_findn(NULL, path, len);
 	if (trans)
 	    res = callback(trans, strlen(trans), arg);
 	else
 	    res = callback(path, len, arg);
 
 	if (res != NULL)
 	    return (res);
 
 	path += len;
 	path += strspn(path, ":;");
     }
 
     return (NULL);
 }
 
 struct try_library_args {
     const char	*name;
     size_t	 namelen;
     char	*buffer;
     size_t	 buflen;
 };
 
 static void *
 try_library_path(const char *dir, size_t dirlen, void *param)
 {
     struct try_library_args *arg;
 
     arg = param;
     if (*dir == '/' || trust) {
 	char *pathname;
 
 	if (dirlen + 1 + arg->namelen + 1 > arg->buflen)
 		return (NULL);
 
 	pathname = arg->buffer;
 	strncpy(pathname, dir, dirlen);
 	pathname[dirlen] = '/';
 	strcpy(pathname + dirlen + 1, arg->name);
 
 	dbg("  Trying \"%s\"", pathname);
 	if (access(pathname, F_OK) == 0) {		/* We found it */
 	    pathname = xmalloc(dirlen + 1 + arg->namelen + 1);
 	    strcpy(pathname, arg->buffer);
 	    return (pathname);
 	}
     }
     return (NULL);
 }
 
 static char *
 search_library_path(const char *name, const char *path)
 {
     char *p;
     struct try_library_args arg;
 
     if (path == NULL)
 	return NULL;
 
     arg.name = name;
     arg.namelen = strlen(name);
     arg.buffer = xmalloc(PATH_MAX);
     arg.buflen = PATH_MAX;
 
     p = path_enumerate(path, try_library_path, &arg);
 
     free(arg.buffer);
 
     return (p);
 }
 
 
 /*
  * Finds the library with the given name using the directory descriptors
  * listed in the LD_LIBRARY_PATH_FDS environment variable.
  *
  * Returns a freshly-opened close-on-exec file descriptor for the library,
  * or -1 if the library cannot be found.
  */
 static char *
 search_library_pathfds(const char *name, const char *path, int *fdp)
 {
 	char *envcopy, *fdstr, *found, *last_token;
 	size_t len;
 	int dirfd, fd;
 
 	dbg("%s('%s', '%s', fdp)", __func__, name, path);
 
 	/* Don't load from user-specified libdirs into setuid binaries. */
 	if (!trust)
 		return (NULL);
 
 	/* We can't do anything if LD_LIBRARY_PATH_FDS isn't set. */
 	if (path == NULL)
 		return (NULL);
 
 	/* LD_LIBRARY_PATH_FDS only works with relative paths. */
 	if (name[0] == '/') {
 		dbg("Absolute path (%s) passed to %s", name, __func__);
 		return (NULL);
 	}
 
 	/*
 	 * Use strtok_r() to walk the FD:FD:FD list.  This requires a local
 	 * copy of the path, as strtok_r rewrites separator tokens
 	 * with '\0'.
 	 */
 	found = NULL;
 	envcopy = xstrdup(path);
 	for (fdstr = strtok_r(envcopy, ":", &last_token); fdstr != NULL;
 	    fdstr = strtok_r(NULL, ":", &last_token)) {
 		dirfd = parse_libdir(fdstr);
 		if (dirfd < 0)
 			break;
 		fd = __sys_openat(dirfd, name, O_RDONLY | O_CLOEXEC);
 		if (fd >= 0) {
 			*fdp = fd;
 			len = strlen(fdstr) + strlen(name) + 3;
 			found = xmalloc(len);
 			if (rtld_snprintf(found, len, "#%d/%s", dirfd, name) < 0) {
 				_rtld_error("error generating '%d/%s'",
 				    dirfd, name);
 				die();
 			}
 			dbg("open('%s') => %d", found, fd);
 			break;
 		}
 	}
 	free(envcopy);
 
 	return (found);
 }
 
 
 int
 dlclose(void *handle)
 {
     Obj_Entry *root;
     RtldLockState lockstate;
 
     wlock_acquire(rtld_bind_lock, &lockstate);
     root = dlcheck(handle);
     if (root == NULL) {
 	lock_release(rtld_bind_lock, &lockstate);
 	return -1;
     }
     LD_UTRACE(UTRACE_DLCLOSE_START, handle, NULL, 0, root->dl_refcount,
 	root->path);
 
     /* Unreference the object and its dependencies. */
     root->dl_refcount--;
 
     if (root->refcount == 1) {
 	/*
 	 * The object will be no longer referenced, so we must unload it.
 	 * First, call the fini functions.
 	 */
 	objlist_call_fini(&list_fini, root, &lockstate);
 
 	unref_dag(root);
 
 	/* Finish cleaning up the newly-unreferenced objects. */
 	GDB_STATE(RT_DELETE,&root->linkmap);
 	unload_object(root);
 	GDB_STATE(RT_CONSISTENT,NULL);
     } else
 	unref_dag(root);
 
     LD_UTRACE(UTRACE_DLCLOSE_STOP, handle, NULL, 0, 0, NULL);
     lock_release(rtld_bind_lock, &lockstate);
     return 0;
 }
 
 char *
 dlerror(void)
 {
     char *msg = error_message;
     error_message = NULL;
     return msg;
 }
 
 /*
  * This function is deprecated and has no effect.
  */
 void
 dllockinit(void *context,
 	   void *(*lock_create)(void *context),
            void (*rlock_acquire)(void *lock),
            void (*wlock_acquire)(void *lock),
            void (*lock_release)(void *lock),
            void (*lock_destroy)(void *lock),
 	   void (*context_destroy)(void *context))
 {
     static void *cur_context;
     static void (*cur_context_destroy)(void *);
 
     /* Just destroy the context from the previous call, if necessary. */
     if (cur_context_destroy != NULL)
 	cur_context_destroy(cur_context);
     cur_context = context;
     cur_context_destroy = context_destroy;
 }
 
 void *
 dlopen(const char *name, int mode)
 {
 
 	return (rtld_dlopen(name, -1, mode));
 }
 
 void *
 fdlopen(int fd, int mode)
 {
 
 	return (rtld_dlopen(NULL, fd, mode));
 }
 
 static void *
 rtld_dlopen(const char *name, int fd, int mode)
 {
     RtldLockState lockstate;
     int lo_flags;
 
     LD_UTRACE(UTRACE_DLOPEN_START, NULL, NULL, 0, mode, name);
     ld_tracing = (mode & RTLD_TRACE) == 0 ? NULL : "1";
     if (ld_tracing != NULL) {
 	rlock_acquire(rtld_bind_lock, &lockstate);
 	if (sigsetjmp(lockstate.env, 0) != 0)
 	    lock_upgrade(rtld_bind_lock, &lockstate);
 	environ = (char **)*get_program_var_addr("environ", &lockstate);
 	lock_release(rtld_bind_lock, &lockstate);
     }
     lo_flags = RTLD_LO_DLOPEN;
     if (mode & RTLD_NODELETE)
 	    lo_flags |= RTLD_LO_NODELETE;
     if (mode & RTLD_NOLOAD)
 	    lo_flags |= RTLD_LO_NOLOAD;
     if (ld_tracing != NULL)
 	    lo_flags |= RTLD_LO_TRACE;
 
     return (dlopen_object(name, fd, obj_main, lo_flags,
       mode & (RTLD_MODEMASK | RTLD_GLOBAL), NULL));
 }
 
 static void
 dlopen_cleanup(Obj_Entry *obj)
 {
 
 	obj->dl_refcount--;
 	unref_dag(obj);
 	if (obj->refcount == 0)
 		unload_object(obj);
 }
 
 static Obj_Entry *
 dlopen_object(const char *name, int fd, Obj_Entry *refobj, int lo_flags,
     int mode, RtldLockState *lockstate)
 {
     Obj_Entry **old_obj_tail;
     Obj_Entry *obj;
     Objlist initlist;
     RtldLockState mlockstate;
     int result;
 
     objlist_init(&initlist);
 
     if (lockstate == NULL && !(lo_flags & RTLD_LO_EARLY)) {
 	wlock_acquire(rtld_bind_lock, &mlockstate);
 	lockstate = &mlockstate;
     }
     GDB_STATE(RT_ADD,NULL);
 
     old_obj_tail = obj_tail;
     obj = NULL;
     if (name == NULL && fd == -1) {
 	obj = obj_main;
 	obj->refcount++;
     } else {
 	obj = load_object(name, fd, refobj, lo_flags);
     }
 
     if (obj) {
 	obj->dl_refcount++;
 	if (mode & RTLD_GLOBAL && objlist_find(&list_global, obj) == NULL)
 	    objlist_push_tail(&list_global, obj);
 	if (*old_obj_tail != NULL) {		/* We loaded something new. */
 	    assert(*old_obj_tail == obj);
 	    result = load_needed_objects(obj,
 		lo_flags & (RTLD_LO_DLOPEN | RTLD_LO_EARLY));
 	    init_dag(obj);
 	    ref_dag(obj);
 	    if (result != -1)
 		result = rtld_verify_versions(&obj->dagmembers);
 	    if (result != -1 && ld_tracing)
 		goto trace;
 	    if (result == -1 || relocate_object_dag(obj,
 	      (mode & RTLD_MODEMASK) == RTLD_NOW, &obj_rtld,
 	      (lo_flags & RTLD_LO_EARLY) ? SYMLOOK_EARLY : 0,
 	      lockstate) == -1) {
 		dlopen_cleanup(obj);
 		obj = NULL;
 	    } else if (lo_flags & RTLD_LO_EARLY) {
 		/*
 		 * Do not call the init functions for early loaded
 		 * filtees.  The image is still not initialized enough
 		 * for them to work.
 		 *
 		 * Our object is found by the global object list and
 		 * will be ordered among all init calls done right
 		 * before transferring control to main.
 		 */
 	    } else {
 		/* Make list of init functions to call. */
 		initlist_add_objects(obj, &obj->next, &initlist);
 	    }
 	    /*
 	     * Process all no_delete objects here, given them own
 	     * DAGs to prevent their dependencies from being unloaded.
 	     * This has to be done after we have loaded all of the
 	     * dependencies, so that we do not miss any.
 	     */
 	    if (obj != NULL)
 		process_nodelete(obj);
 	} else {
 	    /*
 	     * Bump the reference counts for objects on this DAG.  If
 	     * this is the first dlopen() call for the object that was
 	     * already loaded as a dependency, initialize the dag
 	     * starting at it.
 	     */
 	    init_dag(obj);
 	    ref_dag(obj);
 
 	    if ((lo_flags & RTLD_LO_TRACE) != 0)
 		goto trace;
 	}
 	if (obj != NULL && ((lo_flags & RTLD_LO_NODELETE) != 0 ||
 	  obj->z_nodelete) && !obj->ref_nodel) {
 	    dbg("obj %s nodelete", obj->path);
 	    ref_dag(obj);
 	    obj->z_nodelete = obj->ref_nodel = true;
 	}
     }
 
     LD_UTRACE(UTRACE_DLOPEN_STOP, obj, NULL, 0, obj ? obj->dl_refcount : 0,
 	name);
     GDB_STATE(RT_CONSISTENT,obj ? &obj->linkmap : NULL);
 
     if (!(lo_flags & RTLD_LO_EARLY)) {
 	map_stacks_exec(lockstate);
     }
 
     if (initlist_objects_ifunc(&initlist, (mode & RTLD_MODEMASK) == RTLD_NOW,
       (lo_flags & RTLD_LO_EARLY) ? SYMLOOK_EARLY : 0,
       lockstate) == -1) {
 	objlist_clear(&initlist);
 	dlopen_cleanup(obj);
 	if (lockstate == &mlockstate)
 	    lock_release(rtld_bind_lock, lockstate);
 	return (NULL);
     }
 
     if (!(lo_flags & RTLD_LO_EARLY)) {
 	/* Call the init functions. */
 	objlist_call_init(&initlist, lockstate);
     }
     objlist_clear(&initlist);
     if (lockstate == &mlockstate)
 	lock_release(rtld_bind_lock, lockstate);
     return obj;
 trace:
     trace_loaded_objects(obj);
     if (lockstate == &mlockstate)
 	lock_release(rtld_bind_lock, lockstate);
     exit(0);
 }
 
 static void *
 do_dlsym(void *handle, const char *name, void *retaddr, const Ver_Entry *ve,
     int flags)
 {
     DoneList donelist;
     const Obj_Entry *obj, *defobj;
     const Elf_Sym *def;
     SymLook req;
     RtldLockState lockstate;
     tls_index ti;
     void *sym;
     int res;
 
     def = NULL;
     defobj = NULL;
     symlook_init(&req, name);
     req.ventry = ve;
     req.flags = flags | SYMLOOK_IN_PLT;
     req.lockstate = &lockstate;
 
     LD_UTRACE(UTRACE_DLSYM_START, handle, NULL, 0, 0, name);
     rlock_acquire(rtld_bind_lock, &lockstate);
     if (sigsetjmp(lockstate.env, 0) != 0)
 	    lock_upgrade(rtld_bind_lock, &lockstate);
     if (handle == NULL || handle == RTLD_NEXT ||
 	handle == RTLD_DEFAULT || handle == RTLD_SELF) {
 
 	if ((obj = obj_from_addr(retaddr)) == NULL) {
 	    _rtld_error("Cannot determine caller's shared object");
 	    lock_release(rtld_bind_lock, &lockstate);
 	    LD_UTRACE(UTRACE_DLSYM_STOP, handle, NULL, 0, 0, name);
 	    return NULL;
 	}
 	if (handle == NULL) {	/* Just the caller's shared object. */
 	    res = symlook_obj(&req, obj);
 	    if (res == 0) {
 		def = req.sym_out;
 		defobj = req.defobj_out;
 	    }
 	} else if (handle == RTLD_NEXT || /* Objects after caller's */
 		   handle == RTLD_SELF) { /* ... caller included */
 	    if (handle == RTLD_NEXT)
 		obj = obj->next;
 	    for (; obj != NULL; obj = obj->next) {
 		res = symlook_obj(&req, obj);
 		if (res == 0) {
 		    if (def == NULL ||
 		      ELF_ST_BIND(req.sym_out->st_info) != STB_WEAK) {
 			def = req.sym_out;
 			defobj = req.defobj_out;
 			if (ELF_ST_BIND(def->st_info) != STB_WEAK)
 			    break;
 		    }
 		}
 	    }
 	    /*
 	     * Search the dynamic linker itself, and possibly resolve the
 	     * symbol from there.  This is how the application links to
 	     * dynamic linker services such as dlopen.
 	     */
 	    if (def == NULL || ELF_ST_BIND(def->st_info) == STB_WEAK) {
 		res = symlook_obj(&req, &obj_rtld);
 		if (res == 0) {
 		    def = req.sym_out;
 		    defobj = req.defobj_out;
 		}
 	    }
 	} else {
 	    assert(handle == RTLD_DEFAULT);
 	    res = symlook_default(&req, obj);
 	    if (res == 0) {
 		defobj = req.defobj_out;
 		def = req.sym_out;
 	    }
 	}
     } else {
 	if ((obj = dlcheck(handle)) == NULL) {
 	    lock_release(rtld_bind_lock, &lockstate);
 	    LD_UTRACE(UTRACE_DLSYM_STOP, handle, NULL, 0, 0, name);
 	    return NULL;
 	}
 
 	donelist_init(&donelist);
 	if (obj->mainprog) {
             /* Handle obtained by dlopen(NULL, ...) implies global scope. */
 	    res = symlook_global(&req, &donelist);
 	    if (res == 0) {
 		def = req.sym_out;
 		defobj = req.defobj_out;
 	    }
 	    /*
 	     * Search the dynamic linker itself, and possibly resolve the
 	     * symbol from there.  This is how the application links to
 	     * dynamic linker services such as dlopen.
 	     */
 	    if (def == NULL || ELF_ST_BIND(def->st_info) == STB_WEAK) {
 		res = symlook_obj(&req, &obj_rtld);
 		if (res == 0) {
 		    def = req.sym_out;
 		    defobj = req.defobj_out;
 		}
 	    }
 	}
 	else {
 	    /* Search the whole DAG rooted at the given object. */
 	    res = symlook_list(&req, &obj->dagmembers, &donelist);
 	    if (res == 0) {
 		def = req.sym_out;
 		defobj = req.defobj_out;
 	    }
 	}
     }
 
     if (def != NULL) {
 	lock_release(rtld_bind_lock, &lockstate);
 
 	/*
 	 * The value required by the caller is derived from the value
 	 * of the symbol. this is simply the relocated value of the
 	 * symbol.
 	 */
 	if (ELF_ST_TYPE(def->st_info) == STT_FUNC)
 	    sym = make_function_pointer(def, defobj);
 	else if (ELF_ST_TYPE(def->st_info) == STT_GNU_IFUNC)
 	    sym = rtld_resolve_ifunc(defobj, def);
 	else if (ELF_ST_TYPE(def->st_info) == STT_TLS) {
 	    ti.ti_module = defobj->tlsindex;
 	    ti.ti_offset = def->st_value;
 	    sym = __tls_get_addr(&ti);
 	} else
 	    sym = defobj->relocbase + def->st_value;
 	LD_UTRACE(UTRACE_DLSYM_STOP, handle, sym, 0, 0, name);
 	return (sym);
     }
 
     _rtld_error("Undefined symbol \"%s\"", name);
     lock_release(rtld_bind_lock, &lockstate);
     LD_UTRACE(UTRACE_DLSYM_STOP, handle, NULL, 0, 0, name);
     return NULL;
 }
 
 void *
 dlsym(void *handle, const char *name)
 {
 	return do_dlsym(handle, name, __builtin_return_address(0), NULL,
 	    SYMLOOK_DLSYM);
 }
 
 dlfunc_t
 dlfunc(void *handle, const char *name)
 {
 	union {
 		void *d;
 		dlfunc_t f;
 	} rv;
 
 	rv.d = do_dlsym(handle, name, __builtin_return_address(0), NULL,
 	    SYMLOOK_DLSYM);
 	return (rv.f);
 }
 
 void *
 dlvsym(void *handle, const char *name, const char *version)
 {
 	Ver_Entry ventry;
 
 	ventry.name = version;
 	ventry.file = NULL;
 	ventry.hash = elf_hash(version);
 	ventry.flags= 0;
 	return do_dlsym(handle, name, __builtin_return_address(0), &ventry,
 	    SYMLOOK_DLSYM);
 }
 
 int
 _rtld_addr_phdr(const void *addr, struct dl_phdr_info *phdr_info)
 {
     const Obj_Entry *obj;
     RtldLockState lockstate;
 
     rlock_acquire(rtld_bind_lock, &lockstate);
     obj = obj_from_addr(addr);
     if (obj == NULL) {
         _rtld_error("No shared object contains address");
 	lock_release(rtld_bind_lock, &lockstate);
         return (0);
     }
     rtld_fill_dl_phdr_info(obj, phdr_info);
     lock_release(rtld_bind_lock, &lockstate);
     return (1);
 }
 
 int
 dladdr(const void *addr, Dl_info *info)
 {
     const Obj_Entry *obj;
     const Elf_Sym *def;
     void *symbol_addr;
     unsigned long symoffset;
     RtldLockState lockstate;
 
     rlock_acquire(rtld_bind_lock, &lockstate);
     obj = obj_from_addr(addr);
     if (obj == NULL) {
         _rtld_error("No shared object contains address");
 	lock_release(rtld_bind_lock, &lockstate);
         return 0;
     }
     info->dli_fname = obj->path;
     info->dli_fbase = obj->mapbase;
     info->dli_saddr = (void *)0;
     info->dli_sname = NULL;
 
     /*
      * Walk the symbol list looking for the symbol whose address is
      * closest to the address sent in.
      */
     for (symoffset = 0; symoffset < obj->dynsymcount; symoffset++) {
         def = obj->symtab + symoffset;
 
         /*
          * For skip the symbol if st_shndx is either SHN_UNDEF or
          * SHN_COMMON.
          */
         if (def->st_shndx == SHN_UNDEF || def->st_shndx == SHN_COMMON)
             continue;
 
         /*
          * If the symbol is greater than the specified address, or if it
          * is further away from addr than the current nearest symbol,
          * then reject it.
          */
         symbol_addr = obj->relocbase + def->st_value;
         if (symbol_addr > addr || symbol_addr < info->dli_saddr)
             continue;
 
         /* Update our idea of the nearest symbol. */
         info->dli_sname = obj->strtab + def->st_name;
         info->dli_saddr = symbol_addr;
 
         /* Exact match? */
         if (info->dli_saddr == addr)
             break;
     }
     lock_release(rtld_bind_lock, &lockstate);
     return 1;
 }
 
 int
 dlinfo(void *handle, int request, void *p)
 {
     const Obj_Entry *obj;
     RtldLockState lockstate;
     int error;
 
     rlock_acquire(rtld_bind_lock, &lockstate);
 
     if (handle == NULL || handle == RTLD_SELF) {
 	void *retaddr;
 
 	retaddr = __builtin_return_address(0);	/* __GNUC__ only */
 	if ((obj = obj_from_addr(retaddr)) == NULL)
 	    _rtld_error("Cannot determine caller's shared object");
     } else
 	obj = dlcheck(handle);
 
     if (obj == NULL) {
 	lock_release(rtld_bind_lock, &lockstate);
 	return (-1);
     }
 
     error = 0;
     switch (request) {
     case RTLD_DI_LINKMAP:
 	*((struct link_map const **)p) = &obj->linkmap;
 	break;
     case RTLD_DI_ORIGIN:
 	error = rtld_dirname(obj->path, p);
 	break;
 
     case RTLD_DI_SERINFOSIZE:
     case RTLD_DI_SERINFO:
 	error = do_search_info(obj, request, (struct dl_serinfo *)p);
 	break;
 
     default:
 	_rtld_error("Invalid request %d passed to dlinfo()", request);
 	error = -1;
     }
 
     lock_release(rtld_bind_lock, &lockstate);
 
     return (error);
 }
 
 static void
 rtld_fill_dl_phdr_info(const Obj_Entry *obj, struct dl_phdr_info *phdr_info)
 {
 
 	phdr_info->dlpi_addr = (Elf_Addr)obj->relocbase;
 	phdr_info->dlpi_name = obj->path;
 	phdr_info->dlpi_phdr = obj->phdr;
 	phdr_info->dlpi_phnum = obj->phsize / sizeof(obj->phdr[0]);
 	phdr_info->dlpi_tls_modid = obj->tlsindex;
 	phdr_info->dlpi_tls_data = obj->tlsinit;
 	phdr_info->dlpi_adds = obj_loads;
 	phdr_info->dlpi_subs = obj_loads - obj_count;
 }
 
 int
 dl_iterate_phdr(__dl_iterate_hdr_callback callback, void *param)
 {
     struct dl_phdr_info phdr_info;
     const Obj_Entry *obj;
     RtldLockState bind_lockstate, phdr_lockstate;
     int error;
 
     wlock_acquire(rtld_phdr_lock, &phdr_lockstate);
     rlock_acquire(rtld_bind_lock, &bind_lockstate);
 
     error = 0;
 
     for (obj = obj_list;  obj != NULL;  obj = obj->next) {
 	rtld_fill_dl_phdr_info(obj, &phdr_info);
 	if ((error = callback(&phdr_info, sizeof phdr_info, param)) != 0)
 		break;
 
     }
     if (error == 0) {
 	rtld_fill_dl_phdr_info(&obj_rtld, &phdr_info);
 	error = callback(&phdr_info, sizeof(phdr_info), param);
     }
 
     lock_release(rtld_bind_lock, &bind_lockstate);
     lock_release(rtld_phdr_lock, &phdr_lockstate);
 
     return (error);
 }
 
 static void *
 fill_search_info(const char *dir, size_t dirlen, void *param)
 {
     struct fill_search_info_args *arg;
 
     arg = param;
 
     if (arg->request == RTLD_DI_SERINFOSIZE) {
 	arg->serinfo->dls_cnt ++;
 	arg->serinfo->dls_size += sizeof(struct dl_serpath) + dirlen + 1;
     } else {
 	struct dl_serpath *s_entry;
 
 	s_entry = arg->serpath;
 	s_entry->dls_name  = arg->strspace;
 	s_entry->dls_flags = arg->flags;
 
 	strncpy(arg->strspace, dir, dirlen);
 	arg->strspace[dirlen] = '\0';
 
 	arg->strspace += dirlen + 1;
 	arg->serpath++;
     }
 
     return (NULL);
 }
 
 static int
 do_search_info(const Obj_Entry *obj, int request, struct dl_serinfo *info)
 {
     struct dl_serinfo _info;
     struct fill_search_info_args args;
 
     args.request = RTLD_DI_SERINFOSIZE;
     args.serinfo = &_info;
 
     _info.dls_size = __offsetof(struct dl_serinfo, dls_serpath);
     _info.dls_cnt  = 0;
 
     path_enumerate(obj->rpath, fill_search_info, &args);
     path_enumerate(ld_library_path, fill_search_info, &args);
     path_enumerate(obj->runpath, fill_search_info, &args);
     path_enumerate(gethints(obj->z_nodeflib), fill_search_info, &args);
     if (!obj->z_nodeflib)
       path_enumerate(STANDARD_LIBRARY_PATH, fill_search_info, &args);
 
 
     if (request == RTLD_DI_SERINFOSIZE) {
 	info->dls_size = _info.dls_size;
 	info->dls_cnt = _info.dls_cnt;
 	return (0);
     }
 
     if (info->dls_cnt != _info.dls_cnt || info->dls_size != _info.dls_size) {
 	_rtld_error("Uninitialized Dl_serinfo struct passed to dlinfo()");
 	return (-1);
     }
 
     args.request  = RTLD_DI_SERINFO;
     args.serinfo  = info;
     args.serpath  = &info->dls_serpath[0];
     args.strspace = (char *)&info->dls_serpath[_info.dls_cnt];
 
     args.flags = LA_SER_RUNPATH;
     if (path_enumerate(obj->rpath, fill_search_info, &args) != NULL)
 	return (-1);
 
     args.flags = LA_SER_LIBPATH;
     if (path_enumerate(ld_library_path, fill_search_info, &args) != NULL)
 	return (-1);
 
     args.flags = LA_SER_RUNPATH;
     if (path_enumerate(obj->runpath, fill_search_info, &args) != NULL)
 	return (-1);
 
     args.flags = LA_SER_CONFIG;
     if (path_enumerate(gethints(obj->z_nodeflib), fill_search_info, &args)
       != NULL)
 	return (-1);
 
     args.flags = LA_SER_DEFAULT;
     if (!obj->z_nodeflib &&
       path_enumerate(STANDARD_LIBRARY_PATH, fill_search_info, &args) != NULL)
 	return (-1);
     return (0);
 }
 
 static int
 rtld_dirname(const char *path, char *bname)
 {
     const char *endp;
 
     /* Empty or NULL string gets treated as "." */
     if (path == NULL || *path == '\0') {
 	bname[0] = '.';
 	bname[1] = '\0';
 	return (0);
     }
 
     /* Strip trailing slashes */
     endp = path + strlen(path) - 1;
     while (endp > path && *endp == '/')
 	endp--;
 
     /* Find the start of the dir */
     while (endp > path && *endp != '/')
 	endp--;
 
     /* Either the dir is "/" or there are no slashes */
     if (endp == path) {
 	bname[0] = *endp == '/' ? '/' : '.';
 	bname[1] = '\0';
 	return (0);
     } else {
 	do {
 	    endp--;
 	} while (endp > path && *endp == '/');
     }
 
     if (endp - path + 2 > PATH_MAX)
     {
 	_rtld_error("Filename is too long: %s", path);
 	return(-1);
     }
 
     strncpy(bname, path, endp - path + 1);
     bname[endp - path + 1] = '\0';
     return (0);
 }
 
 static int
 rtld_dirname_abs(const char *path, char *base)
 {
 	char base_rel[PATH_MAX];
 
 	if (rtld_dirname(path, base) == -1)
 		return (-1);
 	if (base[0] == '/')
 		return (0);
 	if (getcwd(base_rel, sizeof(base_rel)) == NULL ||
 	    strlcat(base_rel, "/", sizeof(base_rel)) >= sizeof(base_rel) ||
 	    strlcat(base_rel, base, sizeof(base_rel)) >= sizeof(base_rel))
 		return (-1);
 	strcpy(base, base_rel);
 	return (0);
 }
 
 static void
 linkmap_add(Obj_Entry *obj)
 {
     struct link_map *l = &obj->linkmap;
     struct link_map *prev;
 
     obj->linkmap.l_name = obj->path;
     obj->linkmap.l_addr = obj->mapbase;
     obj->linkmap.l_ld = obj->dynamic;
 #ifdef __mips__
     /* GDB needs load offset on MIPS to use the symbols */
     obj->linkmap.l_offs = obj->relocbase;
 #endif
 
     if (r_debug.r_map == NULL) {
 	r_debug.r_map = l;
 	return;
     }
 
     /*
      * Scan to the end of the list, but not past the entry for the
      * dynamic linker, which we want to keep at the very end.
      */
     for (prev = r_debug.r_map;
       prev->l_next != NULL && prev->l_next != &obj_rtld.linkmap;
       prev = prev->l_next)
 	;
 
     /* Link in the new entry. */
     l->l_prev = prev;
     l->l_next = prev->l_next;
     if (l->l_next != NULL)
 	l->l_next->l_prev = l;
     prev->l_next = l;
 }
 
 static void
 linkmap_delete(Obj_Entry *obj)
 {
     struct link_map *l = &obj->linkmap;
 
     if (l->l_prev == NULL) {
 	if ((r_debug.r_map = l->l_next) != NULL)
 	    l->l_next->l_prev = NULL;
 	return;
     }
 
     if ((l->l_prev->l_next = l->l_next) != NULL)
 	l->l_next->l_prev = l->l_prev;
 }
 
 /*
  * Function for the debugger to set a breakpoint on to gain control.
  *
  * The two parameters allow the debugger to easily find and determine
  * what the runtime loader is doing and to whom it is doing it.
  *
  * When the loadhook trap is hit (r_debug_state, set at program
  * initialization), the arguments can be found on the stack:
  *
  *  +8   struct link_map *m
  *  +4   struct r_debug  *rd
  *  +0   RetAddr
  */
 void
 r_debug_state(struct r_debug* rd, struct link_map *m)
 {
     /*
      * The following is a hack to force the compiler to emit calls to
      * this function, even when optimizing.  If the function is empty,
      * the compiler is not obliged to emit any code for calls to it,
      * even when marked __noinline.  However, gdb depends on those
      * calls being made.
      */
     __compiler_membar();
 }
 
 /*
  * A function called after init routines have completed. This can be used to
  * break before a program's entry routine is called, and can be used when
  * main is not available in the symbol table.
  */
 void
 _r_debug_postinit(struct link_map *m)
 {
 
 	/* See r_debug_state(). */
 	__compiler_membar();
 }
 
 /*
  * Get address of the pointer variable in the main program.
  * Prefer non-weak symbol over the weak one.
  */
 static const void **
 get_program_var_addr(const char *name, RtldLockState *lockstate)
 {
     SymLook req;
     DoneList donelist;
 
     symlook_init(&req, name);
     req.lockstate = lockstate;
     donelist_init(&donelist);
     if (symlook_global(&req, &donelist) != 0)
 	return (NULL);
     if (ELF_ST_TYPE(req.sym_out->st_info) == STT_FUNC)
 	return ((const void **)make_function_pointer(req.sym_out,
 	  req.defobj_out));
     else if (ELF_ST_TYPE(req.sym_out->st_info) == STT_GNU_IFUNC)
 	return ((const void **)rtld_resolve_ifunc(req.defobj_out, req.sym_out));
     else
 	return ((const void **)(req.defobj_out->relocbase +
 	  req.sym_out->st_value));
 }
 
 /*
  * Set a pointer variable in the main program to the given value.  This
  * is used to set key variables such as "environ" before any of the
  * init functions are called.
  */
 static void
 set_program_var(const char *name, const void *value)
 {
     const void **addr;
 
     if ((addr = get_program_var_addr(name, NULL)) != NULL) {
 	dbg("\"%s\": *%p <-- %p", name, addr, value);
 	*addr = value;
     }
 }
 
 /*
  * Search the global objects, including dependencies and main object,
  * for the given symbol.
  */
 static int
 symlook_global(SymLook *req, DoneList *donelist)
 {
     SymLook req1;
     const Objlist_Entry *elm;
     int res;
 
     symlook_init_from_req(&req1, req);
 
     /* Search all objects loaded at program start up. */
     if (req->defobj_out == NULL ||
       ELF_ST_BIND(req->sym_out->st_info) == STB_WEAK) {
 	res = symlook_list(&req1, &list_main, donelist);
 	if (res == 0 && (req->defobj_out == NULL ||
 	  ELF_ST_BIND(req1.sym_out->st_info) != STB_WEAK)) {
 	    req->sym_out = req1.sym_out;
 	    req->defobj_out = req1.defobj_out;
 	    assert(req->defobj_out != NULL);
 	}
     }
 
     /* Search all DAGs whose roots are RTLD_GLOBAL objects. */
     STAILQ_FOREACH(elm, &list_global, link) {
 	if (req->defobj_out != NULL &&
 	  ELF_ST_BIND(req->sym_out->st_info) != STB_WEAK)
 	    break;
 	res = symlook_list(&req1, &elm->obj->dagmembers, donelist);
 	if (res == 0 && (req->defobj_out == NULL ||
 	  ELF_ST_BIND(req1.sym_out->st_info) != STB_WEAK)) {
 	    req->sym_out = req1.sym_out;
 	    req->defobj_out = req1.defobj_out;
 	    assert(req->defobj_out != NULL);
 	}
     }
 
     return (req->sym_out != NULL ? 0 : ESRCH);
 }
 
 /*
  * Given a symbol name in a referencing object, find the corresponding
  * definition of the symbol.  Returns a pointer to the symbol, or NULL if
  * no definition was found.  Returns a pointer to the Obj_Entry of the
  * defining object via the reference parameter DEFOBJ_OUT.
  */
 static int
 symlook_default(SymLook *req, const Obj_Entry *refobj)
 {
     DoneList donelist;
     const Objlist_Entry *elm;
     SymLook req1;
     int res;
 
     donelist_init(&donelist);
     symlook_init_from_req(&req1, req);
 
     /* Look first in the referencing object if linked symbolically. */
     if (refobj->symbolic && !donelist_check(&donelist, refobj)) {
 	res = symlook_obj(&req1, refobj);
 	if (res == 0) {
 	    req->sym_out = req1.sym_out;
 	    req->defobj_out = req1.defobj_out;
 	    assert(req->defobj_out != NULL);
 	}
     }
 
     symlook_global(req, &donelist);
 
     /* Search all dlopened DAGs containing the referencing object. */
     STAILQ_FOREACH(elm, &refobj->dldags, link) {
 	if (req->sym_out != NULL &&
 	  ELF_ST_BIND(req->sym_out->st_info) != STB_WEAK)
 	    break;
 	res = symlook_list(&req1, &elm->obj->dagmembers, &donelist);
 	if (res == 0 && (req->sym_out == NULL ||
 	  ELF_ST_BIND(req1.sym_out->st_info) != STB_WEAK)) {
 	    req->sym_out = req1.sym_out;
 	    req->defobj_out = req1.defobj_out;
 	    assert(req->defobj_out != NULL);
 	}
     }
 
     /*
      * Search the dynamic linker itself, and possibly resolve the
      * symbol from there.  This is how the application links to
      * dynamic linker services such as dlopen.
      */
     if (req->sym_out == NULL ||
       ELF_ST_BIND(req->sym_out->st_info) == STB_WEAK) {
 	res = symlook_obj(&req1, &obj_rtld);
 	if (res == 0) {
 	    req->sym_out = req1.sym_out;
 	    req->defobj_out = req1.defobj_out;
 	    assert(req->defobj_out != NULL);
 	}
     }
 
     return (req->sym_out != NULL ? 0 : ESRCH);
 }
 
 static int
 symlook_list(SymLook *req, const Objlist *objlist, DoneList *dlp)
 {
     const Elf_Sym *def;
     const Obj_Entry *defobj;
     const Objlist_Entry *elm;
     SymLook req1;
     int res;
 
     def = NULL;
     defobj = NULL;
     STAILQ_FOREACH(elm, objlist, link) {
 	if (donelist_check(dlp, elm->obj))
 	    continue;
 	symlook_init_from_req(&req1, req);
 	if ((res = symlook_obj(&req1, elm->obj)) == 0) {
 	    if (def == NULL || ELF_ST_BIND(req1.sym_out->st_info) != STB_WEAK) {
 		def = req1.sym_out;
 		defobj = req1.defobj_out;
 		if (ELF_ST_BIND(def->st_info) != STB_WEAK)
 		    break;
 	    }
 	}
     }
     if (def != NULL) {
 	req->sym_out = def;
 	req->defobj_out = defobj;
 	return (0);
     }
     return (ESRCH);
 }
 
 /*
  * Search the chain of DAGS cointed to by the given Needed_Entry
  * for a symbol of the given name.  Each DAG is scanned completely
  * before advancing to the next one.  Returns a pointer to the symbol,
  * or NULL if no definition was found.
  */
 static int
 symlook_needed(SymLook *req, const Needed_Entry *needed, DoneList *dlp)
 {
     const Elf_Sym *def;
     const Needed_Entry *n;
     const Obj_Entry *defobj;
     SymLook req1;
     int res;
 
     def = NULL;
     defobj = NULL;
     symlook_init_from_req(&req1, req);
     for (n = needed; n != NULL; n = n->next) {
 	if (n->obj == NULL ||
 	    (res = symlook_list(&req1, &n->obj->dagmembers, dlp)) != 0)
 	    continue;
 	if (def == NULL || ELF_ST_BIND(req1.sym_out->st_info) != STB_WEAK) {
 	    def = req1.sym_out;
 	    defobj = req1.defobj_out;
 	    if (ELF_ST_BIND(def->st_info) != STB_WEAK)
 		break;
 	}
     }
     if (def != NULL) {
 	req->sym_out = def;
 	req->defobj_out = defobj;
 	return (0);
     }
     return (ESRCH);
 }
 
 /*
  * Search the symbol table of a single shared object for a symbol of
  * the given name and version, if requested.  Returns a pointer to the
  * symbol, or NULL if no definition was found.  If the object is
  * filter, return filtered symbol from filtee.
  *
  * The symbol's hash value is passed in for efficiency reasons; that
  * eliminates many recomputations of the hash value.
  */
 int
 symlook_obj(SymLook *req, const Obj_Entry *obj)
 {
     DoneList donelist;
     SymLook req1;
     int flags, res, mres;
 
     /*
      * If there is at least one valid hash at this point, we prefer to
      * use the faster GNU version if available.
      */
     if (obj->valid_hash_gnu)
 	mres = symlook_obj1_gnu(req, obj);
     else if (obj->valid_hash_sysv)
 	mres = symlook_obj1_sysv(req, obj);
     else
 	return (EINVAL);
 
     if (mres == 0) {
 	if (obj->needed_filtees != NULL) {
 	    flags = (req->flags & SYMLOOK_EARLY) ? RTLD_LO_EARLY : 0;
 	    load_filtees(__DECONST(Obj_Entry *, obj), flags, req->lockstate);
 	    donelist_init(&donelist);
 	    symlook_init_from_req(&req1, req);
 	    res = symlook_needed(&req1, obj->needed_filtees, &donelist);
 	    if (res == 0) {
 		req->sym_out = req1.sym_out;
 		req->defobj_out = req1.defobj_out;
 	    }
 	    return (res);
 	}
 	if (obj->needed_aux_filtees != NULL) {
 	    flags = (req->flags & SYMLOOK_EARLY) ? RTLD_LO_EARLY : 0;
 	    load_filtees(__DECONST(Obj_Entry *, obj), flags, req->lockstate);
 	    donelist_init(&donelist);
 	    symlook_init_from_req(&req1, req);
 	    res = symlook_needed(&req1, obj->needed_aux_filtees, &donelist);
 	    if (res == 0) {
 		req->sym_out = req1.sym_out;
 		req->defobj_out = req1.defobj_out;
 		return (res);
 	    }
 	}
     }
     return (mres);
 }
 
 /* Symbol match routine common to both hash functions */
 static bool
 matched_symbol(SymLook *req, const Obj_Entry *obj, Sym_Match_Result *result,
     const unsigned long symnum)
 {
 	Elf_Versym verndx;
 	const Elf_Sym *symp;
 	const char *strp;
 
 	symp = obj->symtab + symnum;
 	strp = obj->strtab + symp->st_name;
 
 	switch (ELF_ST_TYPE(symp->st_info)) {
 	case STT_FUNC:
 	case STT_NOTYPE:
 	case STT_OBJECT:
 	case STT_COMMON:
 	case STT_GNU_IFUNC:
 		if (symp->st_value == 0)
 			return (false);
 		/* fallthrough */
 	case STT_TLS:
 		if (symp->st_shndx != SHN_UNDEF)
 			break;
 #ifndef __mips__
 		else if (((req->flags & SYMLOOK_IN_PLT) == 0) &&
 		    (ELF_ST_TYPE(symp->st_info) == STT_FUNC))
 			break;
 		/* fallthrough */
 #endif
 	default:
 		return (false);
 	}
 	if (req->name[0] != strp[0] || strcmp(req->name, strp) != 0)
 		return (false);
 
 	if (req->ventry == NULL) {
 		if (obj->versyms != NULL) {
 			verndx = VER_NDX(obj->versyms[symnum]);
 			if (verndx > obj->vernum) {
 				_rtld_error(
 				    "%s: symbol %s references wrong version %d",
 				    obj->path, obj->strtab + symnum, verndx);
 				return (false);
 			}
 			/*
 			 * If we are not called from dlsym (i.e. this
 			 * is a normal relocation from unversioned
 			 * binary), accept the symbol immediately if
 			 * it happens to have first version after this
 			 * shared object became versioned.  Otherwise,
 			 * if symbol is versioned and not hidden,
 			 * remember it. If it is the only symbol with
 			 * this name exported by the shared object, it
 			 * will be returned as a match by the calling
 			 * function. If symbol is global (verndx < 2)
 			 * accept it unconditionally.
 			 */
 			if ((req->flags & SYMLOOK_DLSYM) == 0 &&
 			    verndx == VER_NDX_GIVEN) {
 				result->sym_out = symp;
 				return (true);
 			}
 			else if (verndx >= VER_NDX_GIVEN) {
 				if ((obj->versyms[symnum] & VER_NDX_HIDDEN)
 				    == 0) {
 					if (result->vsymp == NULL)
 						result->vsymp = symp;
 					result->vcount++;
 				}
 				return (false);
 			}
 		}
 		result->sym_out = symp;
 		return (true);
 	}
 	if (obj->versyms == NULL) {
 		if (object_match_name(obj, req->ventry->name)) {
 			_rtld_error("%s: object %s should provide version %s "
 			    "for symbol %s", obj_rtld.path, obj->path,
 			    req->ventry->name, obj->strtab + symnum);
 			return (false);
 		}
 	} else {
 		verndx = VER_NDX(obj->versyms[symnum]);
 		if (verndx > obj->vernum) {
 			_rtld_error("%s: symbol %s references wrong version %d",
 			    obj->path, obj->strtab + symnum, verndx);
 			return (false);
 		}
 		if (obj->vertab[verndx].hash != req->ventry->hash ||
 		    strcmp(obj->vertab[verndx].name, req->ventry->name)) {
 			/*
 			 * Version does not match. Look if this is a
 			 * global symbol and if it is not hidden. If
 			 * global symbol (verndx < 2) is available,
 			 * use it. Do not return symbol if we are
 			 * called by dlvsym, because dlvsym looks for
 			 * a specific version and default one is not
 			 * what dlvsym wants.
 			 */
 			if ((req->flags & SYMLOOK_DLSYM) ||
 			    (verndx >= VER_NDX_GIVEN) ||
 			    (obj->versyms[symnum] & VER_NDX_HIDDEN))
 				return (false);
 		}
 	}
 	result->sym_out = symp;
 	return (true);
 }
 
 /*
  * Search for symbol using SysV hash function.
  * obj->buckets is known not to be NULL at this point; the test for this was
  * performed with the obj->valid_hash_sysv assignment.
  */
 static int
 symlook_obj1_sysv(SymLook *req, const Obj_Entry *obj)
 {
 	unsigned long symnum;
 	Sym_Match_Result matchres;
 
 	matchres.sym_out = NULL;
 	matchres.vsymp = NULL;
 	matchres.vcount = 0;
 
 	for (symnum = obj->buckets[req->hash % obj->nbuckets];
 	    symnum != STN_UNDEF; symnum = obj->chains[symnum]) {
 		if (symnum >= obj->nchains)
 			return (ESRCH);	/* Bad object */
 
 		if (matched_symbol(req, obj, &matchres, symnum)) {
 			req->sym_out = matchres.sym_out;
 			req->defobj_out = obj;
 			return (0);
 		}
 	}
 	if (matchres.vcount == 1) {
 		req->sym_out = matchres.vsymp;
 		req->defobj_out = obj;
 		return (0);
 	}
 	return (ESRCH);
 }
 
 /* Search for symbol using GNU hash function */
 static int
 symlook_obj1_gnu(SymLook *req, const Obj_Entry *obj)
 {
 	Elf_Addr bloom_word;
 	const Elf32_Word *hashval;
 	Elf32_Word bucket;
 	Sym_Match_Result matchres;
 	unsigned int h1, h2;
 	unsigned long symnum;
 
 	matchres.sym_out = NULL;
 	matchres.vsymp = NULL;
 	matchres.vcount = 0;
 
 	/* Pick right bitmask word from Bloom filter array */
 	bloom_word = obj->bloom_gnu[(req->hash_gnu / __ELF_WORD_SIZE) &
 	    obj->maskwords_bm_gnu];
 
 	/* Calculate modulus word size of gnu hash and its derivative */
 	h1 = req->hash_gnu & (__ELF_WORD_SIZE - 1);
 	h2 = ((req->hash_gnu >> obj->shift2_gnu) & (__ELF_WORD_SIZE - 1));
 
 	/* Filter out the "definitely not in set" queries */
 	if (((bloom_word >> h1) & (bloom_word >> h2) & 1) == 0)
 		return (ESRCH);
 
 	/* Locate hash chain and corresponding value element*/
 	bucket = obj->buckets_gnu[req->hash_gnu % obj->nbuckets_gnu];
 	if (bucket == 0)
 		return (ESRCH);
 	hashval = &obj->chain_zero_gnu[bucket];
 	do {
 		if (((*hashval ^ req->hash_gnu) >> 1) == 0) {
 			symnum = hashval - obj->chain_zero_gnu;
 			if (matched_symbol(req, obj, &matchres, symnum)) {
 				req->sym_out = matchres.sym_out;
 				req->defobj_out = obj;
 				return (0);
 			}
 		}
 	} while ((*hashval++ & 1) == 0);
 	if (matchres.vcount == 1) {
 		req->sym_out = matchres.vsymp;
 		req->defobj_out = obj;
 		return (0);
 	}
 	return (ESRCH);
 }
 
 static void
 trace_loaded_objects(Obj_Entry *obj)
 {
     char	*fmt1, *fmt2, *fmt, *main_local, *list_containers;
     int		c;
 
     if ((main_local = getenv(LD_ "TRACE_LOADED_OBJECTS_PROGNAME")) == NULL)
 	main_local = "";
 
     if ((fmt1 = getenv(LD_ "TRACE_LOADED_OBJECTS_FMT1")) == NULL)
 	fmt1 = "\t%o => %p (%x)\n";
 
     if ((fmt2 = getenv(LD_ "TRACE_LOADED_OBJECTS_FMT2")) == NULL)
 	fmt2 = "\t%o (%x)\n";
 
     list_containers = getenv(LD_ "TRACE_LOADED_OBJECTS_ALL");
 
     for (; obj; obj = obj->next) {
 	Needed_Entry		*needed;
 	char			*name, *path;
 	bool			is_lib;
 
 	if (list_containers && obj->needed != NULL)
 	    rtld_printf("%s:\n", obj->path);
 	for (needed = obj->needed; needed; needed = needed->next) {
 	    if (needed->obj != NULL) {
 		if (needed->obj->traced && !list_containers)
 		    continue;
 		needed->obj->traced = true;
 		path = needed->obj->path;
 	    } else
 		path = "not found";
 
 	    name = (char *)obj->strtab + needed->name;
 	    is_lib = strncmp(name, "lib", 3) == 0;	/* XXX - bogus */
 
 	    fmt = is_lib ? fmt1 : fmt2;
 	    while ((c = *fmt++) != '\0') {
 		switch (c) {
 		default:
 		    rtld_putchar(c);
 		    continue;
 		case '\\':
 		    switch (c = *fmt) {
 		    case '\0':
 			continue;
 		    case 'n':
 			rtld_putchar('\n');
 			break;
 		    case 't':
 			rtld_putchar('\t');
 			break;
 		    }
 		    break;
 		case '%':
 		    switch (c = *fmt) {
 		    case '\0':
 			continue;
 		    case '%':
 		    default:
 			rtld_putchar(c);
 			break;
 		    case 'A':
 			rtld_putstr(main_local);
 			break;
 		    case 'a':
 			rtld_putstr(obj_main->path);
 			break;
 		    case 'o':
 			rtld_putstr(name);
 			break;
 #if 0
 		    case 'm':
 			rtld_printf("%d", sodp->sod_major);
 			break;
 		    case 'n':
 			rtld_printf("%d", sodp->sod_minor);
 			break;
 #endif
 		    case 'p':
 			rtld_putstr(path);
 			break;
 		    case 'x':
 			rtld_printf("%p", needed->obj ? needed->obj->mapbase :
 			  0);
 			break;
 		    }
 		    break;
 		}
 		++fmt;
 	    }
 	}
     }
 }
 
 /*
  * Unload a dlopened object and its dependencies from memory and from
  * our data structures.  It is assumed that the DAG rooted in the
  * object has already been unreferenced, and that the object has a
  * reference count of 0.
  */
 static void
 unload_object(Obj_Entry *root)
 {
     Obj_Entry *obj;
     Obj_Entry **linkp;
 
     assert(root->refcount == 0);
 
     /*
      * Pass over the DAG removing unreferenced objects from
      * appropriate lists.
      */
     unlink_object(root);
 
     /* Unmap all objects that are no longer referenced. */
     linkp = &obj_list->next;
     while ((obj = *linkp) != NULL) {
 	if (obj->refcount == 0) {
 	    LD_UTRACE(UTRACE_UNLOAD_OBJECT, obj, obj->mapbase, obj->mapsize, 0,
 		obj->path);
 	    dbg("unloading \"%s\"", obj->path);
 	    unload_filtees(root);
 	    munmap(obj->mapbase, obj->mapsize);
 	    linkmap_delete(obj);
 	    *linkp = obj->next;
 	    obj_count--;
 	    obj_free(obj);
 	} else
 	    linkp = &obj->next;
     }
     obj_tail = linkp;
 }
 
 static void
 unlink_object(Obj_Entry *root)
 {
     Objlist_Entry *elm;
 
     if (root->refcount == 0) {
 	/* Remove the object from the RTLD_GLOBAL list. */
 	objlist_remove(&list_global, root);
 
     	/* Remove the object from all objects' DAG lists. */
     	STAILQ_FOREACH(elm, &root->dagmembers, link) {
 	    objlist_remove(&elm->obj->dldags, root);
 	    if (elm->obj != root)
 		unlink_object(elm->obj);
 	}
     }
 }
 
 static void
 ref_dag(Obj_Entry *root)
 {
     Objlist_Entry *elm;
 
     assert(root->dag_inited);
     STAILQ_FOREACH(elm, &root->dagmembers, link)
 	elm->obj->refcount++;
 }
 
 static void
 unref_dag(Obj_Entry *root)
 {
     Objlist_Entry *elm;
 
     assert(root->dag_inited);
     STAILQ_FOREACH(elm, &root->dagmembers, link)
 	elm->obj->refcount--;
 }
 
 /*
  * Common code for MD __tls_get_addr().
  */
 static void *tls_get_addr_slow(Elf_Addr **, int, size_t) __noinline;
 static void *
 tls_get_addr_slow(Elf_Addr **dtvp, int index, size_t offset)
 {
     Elf_Addr *newdtv, *dtv;
     RtldLockState lockstate;
     int to_copy;
 
     dtv = *dtvp;
     /* Check dtv generation in case new modules have arrived */
     if (dtv[0] != tls_dtv_generation) {
 	wlock_acquire(rtld_bind_lock, &lockstate);
 	newdtv = xcalloc(tls_max_index + 2, sizeof(Elf_Addr));
 	to_copy = dtv[1];
 	if (to_copy > tls_max_index)
 	    to_copy = tls_max_index;
 	memcpy(&newdtv[2], &dtv[2], to_copy * sizeof(Elf_Addr));
 	newdtv[0] = tls_dtv_generation;
 	newdtv[1] = tls_max_index;
 	free(dtv);
 	lock_release(rtld_bind_lock, &lockstate);
 	dtv = *dtvp = newdtv;
     }
 
     /* Dynamically allocate module TLS if necessary */
     if (dtv[index + 1] == 0) {
 	/* Signal safe, wlock will block out signals. */
 	wlock_acquire(rtld_bind_lock, &lockstate);
 	if (!dtv[index + 1])
 	    dtv[index + 1] = (Elf_Addr)allocate_module_tls(index);
 	lock_release(rtld_bind_lock, &lockstate);
     }
     return ((void *)(dtv[index + 1] + offset));
 }
 
 void *
 tls_get_addr_common(Elf_Addr **dtvp, int index, size_t offset)
 {
 	Elf_Addr *dtv;
 
 	dtv = *dtvp;
 	/* Check dtv generation in case new modules have arrived */
 	if (__predict_true(dtv[0] == tls_dtv_generation &&
 	    dtv[index + 1] != 0))
 		return ((void *)(dtv[index + 1] + offset));
 	return (tls_get_addr_slow(dtvp, index, offset));
 }
 
 #if defined(__arm__) || defined(__mips__) || defined(__powerpc__)
 
 /*
  * Allocate Static TLS using the Variant I method.
  */
 void *
 allocate_tls(Obj_Entry *objs, void *oldtcb, size_t tcbsize, size_t tcbalign)
 {
     Obj_Entry *obj;
     char *tcb;
     Elf_Addr **tls;
     Elf_Addr *dtv;
     Elf_Addr addr;
     int i;
 
     if (oldtcb != NULL && tcbsize == TLS_TCB_SIZE)
 	return (oldtcb);
 
     assert(tcbsize >= TLS_TCB_SIZE);
     tcb = xcalloc(1, tls_static_space - TLS_TCB_SIZE + tcbsize);
     tls = (Elf_Addr **)(tcb + tcbsize - TLS_TCB_SIZE);
 
     if (oldtcb != NULL) {
 	memcpy(tls, oldtcb, tls_static_space);
 	free(oldtcb);
 
 	/* Adjust the DTV. */
 	dtv = tls[0];
 	for (i = 0; i < dtv[1]; i++) {
 	    if (dtv[i+2] >= (Elf_Addr)oldtcb &&
 		dtv[i+2] < (Elf_Addr)oldtcb + tls_static_space) {
 		dtv[i+2] = dtv[i+2] - (Elf_Addr)oldtcb + (Elf_Addr)tls;
 	    }
 	}
     } else {
 	dtv = xcalloc(tls_max_index + 2, sizeof(Elf_Addr));
 	tls[0] = dtv;
 	dtv[0] = tls_dtv_generation;
 	dtv[1] = tls_max_index;
 
 	for (obj = objs; obj; obj = obj->next) {
 	    if (obj->tlsoffset > 0) {
 		addr = (Elf_Addr)tls + obj->tlsoffset;
 		if (obj->tlsinitsize > 0)
 		    memcpy((void*) addr, obj->tlsinit, obj->tlsinitsize);
 		if (obj->tlssize > obj->tlsinitsize)
 		    memset((void*) (addr + obj->tlsinitsize), 0,
 			   obj->tlssize - obj->tlsinitsize);
 		dtv[obj->tlsindex + 1] = addr;
 	    }
 	}
     }
 
     return (tcb);
 }
 
 void
 free_tls(void *tcb, size_t tcbsize, size_t tcbalign)
 {
     Elf_Addr *dtv;
     Elf_Addr tlsstart, tlsend;
     int dtvsize, i;
 
     assert(tcbsize >= TLS_TCB_SIZE);
 
     tlsstart = (Elf_Addr)tcb + tcbsize - TLS_TCB_SIZE;
     tlsend = tlsstart + tls_static_space;
 
     dtv = *(Elf_Addr **)tlsstart;
     dtvsize = dtv[1];
     for (i = 0; i < dtvsize; i++) {
 	if (dtv[i+2] && (dtv[i+2] < tlsstart || dtv[i+2] >= tlsend)) {
 	    free((void*)dtv[i+2]);
 	}
     }
     free(dtv);
     free(tcb);
 }
 
 #endif
 
 #if defined(__i386__) || defined(__amd64__) || defined(__sparc64__)
 
 /*
  * Allocate Static TLS using the Variant II method.
  */
 void *
 allocate_tls(Obj_Entry *objs, void *oldtls, size_t tcbsize, size_t tcbalign)
 {
     Obj_Entry *obj;
     size_t size, ralign;
     char *tls;
     Elf_Addr *dtv, *olddtv;
     Elf_Addr segbase, oldsegbase, addr;
     int i;
 
     ralign = tcbalign;
     if (tls_static_max_align > ralign)
 	    ralign = tls_static_max_align;
     size = round(tls_static_space, ralign) + round(tcbsize, ralign);
 
     assert(tcbsize >= 2*sizeof(Elf_Addr));
     tls = malloc_aligned(size, ralign);
     dtv = xcalloc(tls_max_index + 2, sizeof(Elf_Addr));
 
     segbase = (Elf_Addr)(tls + round(tls_static_space, ralign));
     ((Elf_Addr*)segbase)[0] = segbase;
     ((Elf_Addr*)segbase)[1] = (Elf_Addr) dtv;
 
     dtv[0] = tls_dtv_generation;
     dtv[1] = tls_max_index;
 
     if (oldtls) {
 	/*
 	 * Copy the static TLS block over whole.
 	 */
 	oldsegbase = (Elf_Addr) oldtls;
 	memcpy((void *)(segbase - tls_static_space),
 	       (const void *)(oldsegbase - tls_static_space),
 	       tls_static_space);
 
 	/*
 	 * If any dynamic TLS blocks have been created tls_get_addr(),
 	 * move them over.
 	 */
 	olddtv = ((Elf_Addr**)oldsegbase)[1];
 	for (i = 0; i < olddtv[1]; i++) {
 	    if (olddtv[i+2] < oldsegbase - size || olddtv[i+2] > oldsegbase) {
 		dtv[i+2] = olddtv[i+2];
 		olddtv[i+2] = 0;
 	    }
 	}
 
 	/*
 	 * We assume that this block was the one we created with
 	 * allocate_initial_tls().
 	 */
 	free_tls(oldtls, 2*sizeof(Elf_Addr), sizeof(Elf_Addr));
     } else {
 	for (obj = objs; obj; obj = obj->next) {
 	    if (obj->tlsoffset) {
 		addr = segbase - obj->tlsoffset;
 		memset((void*) (addr + obj->tlsinitsize),
 		       0, obj->tlssize - obj->tlsinitsize);
 		if (obj->tlsinit)
 		    memcpy((void*) addr, obj->tlsinit, obj->tlsinitsize);
 		dtv[obj->tlsindex + 1] = addr;
 	    }
 	}
     }
 
     return (void*) segbase;
 }
 
 void
 free_tls(void *tls, size_t tcbsize, size_t tcbalign)
 {
     Elf_Addr* dtv;
     size_t size, ralign;
     int dtvsize, i;
     Elf_Addr tlsstart, tlsend;
 
     /*
      * Figure out the size of the initial TLS block so that we can
      * find stuff which ___tls_get_addr() allocated dynamically.
      */
     ralign = tcbalign;
     if (tls_static_max_align > ralign)
 	    ralign = tls_static_max_align;
     size = round(tls_static_space, ralign);
 
     dtv = ((Elf_Addr**)tls)[1];
     dtvsize = dtv[1];
     tlsend = (Elf_Addr) tls;
     tlsstart = tlsend - size;
     for (i = 0; i < dtvsize; i++) {
 	if (dtv[i + 2] != 0 && (dtv[i + 2] < tlsstart || dtv[i + 2] > tlsend)) {
 		free_aligned((void *)dtv[i + 2]);
 	}
     }
 
     free_aligned((void *)tlsstart);
     free((void*) dtv);
 }
 
 #endif
 
 /*
  * Allocate TLS block for module with given index.
  */
 void *
 allocate_module_tls(int index)
 {
     Obj_Entry* obj;
     char* p;
 
     for (obj = obj_list; obj; obj = obj->next) {
 	if (obj->tlsindex == index)
 	    break;
     }
     if (!obj) {
 	_rtld_error("Can't find module with TLS index %d", index);
 	die();
     }
 
     p = malloc_aligned(obj->tlssize, obj->tlsalign);
     memcpy(p, obj->tlsinit, obj->tlsinitsize);
     memset(p + obj->tlsinitsize, 0, obj->tlssize - obj->tlsinitsize);
 
     return p;
 }
 
 bool
 allocate_tls_offset(Obj_Entry *obj)
 {
     size_t off;
 
     if (obj->tls_done)
 	return true;
 
     if (obj->tlssize == 0) {
 	obj->tls_done = true;
 	return true;
     }
 
     if (obj->tlsindex == 1)
 	off = calculate_first_tls_offset(obj->tlssize, obj->tlsalign);
     else
 	off = calculate_tls_offset(tls_last_offset, tls_last_size,
 				   obj->tlssize, obj->tlsalign);
 
     /*
      * If we have already fixed the size of the static TLS block, we
      * must stay within that size. When allocating the static TLS, we
      * leave a small amount of space spare to be used for dynamically
      * loading modules which use static TLS.
      */
     if (tls_static_space != 0) {
 	if (calculate_tls_end(off, obj->tlssize) > tls_static_space)
 	    return false;
     } else if (obj->tlsalign > tls_static_max_align) {
 	    tls_static_max_align = obj->tlsalign;
     }
 
     tls_last_offset = obj->tlsoffset = off;
     tls_last_size = obj->tlssize;
     obj->tls_done = true;
 
     return true;
 }
 
 void
 free_tls_offset(Obj_Entry *obj)
 {
 
     /*
      * If we were the last thing to allocate out of the static TLS
      * block, we give our space back to the 'allocator'. This is a
      * simplistic workaround to allow libGL.so.1 to be loaded and
      * unloaded multiple times.
      */
     if (calculate_tls_end(obj->tlsoffset, obj->tlssize)
 	== calculate_tls_end(tls_last_offset, tls_last_size)) {
 	tls_last_offset -= obj->tlssize;
 	tls_last_size = 0;
     }
 }
 
 void *
 _rtld_allocate_tls(void *oldtls, size_t tcbsize, size_t tcbalign)
 {
     void *ret;
     RtldLockState lockstate;
 
     wlock_acquire(rtld_bind_lock, &lockstate);
     ret = allocate_tls(obj_list, oldtls, tcbsize, tcbalign);
     lock_release(rtld_bind_lock, &lockstate);
     return (ret);
 }
 
 void
 _rtld_free_tls(void *tcb, size_t tcbsize, size_t tcbalign)
 {
     RtldLockState lockstate;
 
     wlock_acquire(rtld_bind_lock, &lockstate);
     free_tls(tcb, tcbsize, tcbalign);
     lock_release(rtld_bind_lock, &lockstate);
 }
 
 static void
 object_add_name(Obj_Entry *obj, const char *name)
 {
     Name_Entry *entry;
     size_t len;
 
     len = strlen(name);
     entry = malloc(sizeof(Name_Entry) + len);
 
     if (entry != NULL) {
 	strcpy(entry->name, name);
 	STAILQ_INSERT_TAIL(&obj->names, entry, link);
     }
 }
 
 static int
 object_match_name(const Obj_Entry *obj, const char *name)
 {
     Name_Entry *entry;
 
     STAILQ_FOREACH(entry, &obj->names, link) {
 	if (strcmp(name, entry->name) == 0)
 	    return (1);
     }
     return (0);
 }
 
 static Obj_Entry *
 locate_dependency(const Obj_Entry *obj, const char *name)
 {
     const Objlist_Entry *entry;
     const Needed_Entry *needed;
 
     STAILQ_FOREACH(entry, &list_main, link) {
 	if (object_match_name(entry->obj, name))
 	    return entry->obj;
     }
 
     for (needed = obj->needed;  needed != NULL;  needed = needed->next) {
 	if (strcmp(obj->strtab + needed->name, name) == 0 ||
 	  (needed->obj != NULL && object_match_name(needed->obj, name))) {
 	    /*
 	     * If there is DT_NEEDED for the name we are looking for,
 	     * we are all set.  Note that object might not be found if
 	     * dependency was not loaded yet, so the function can
 	     * return NULL here.  This is expected and handled
 	     * properly by the caller.
 	     */
 	    return (needed->obj);
 	}
     }
     _rtld_error("%s: Unexpected inconsistency: dependency %s not found",
 	obj->path, name);
     die();
 }
 
 static int
 check_object_provided_version(Obj_Entry *refobj, const Obj_Entry *depobj,
     const Elf_Vernaux *vna)
 {
     const Elf_Verdef *vd;
     const char *vername;
 
     vername = refobj->strtab + vna->vna_name;
     vd = depobj->verdef;
     if (vd == NULL) {
 	_rtld_error("%s: version %s required by %s not defined",
 	    depobj->path, vername, refobj->path);
 	return (-1);
     }
     for (;;) {
 	if (vd->vd_version != VER_DEF_CURRENT) {
 	    _rtld_error("%s: Unsupported version %d of Elf_Verdef entry",
 		depobj->path, vd->vd_version);
 	    return (-1);
 	}
 	if (vna->vna_hash == vd->vd_hash) {
 	    const Elf_Verdaux *aux = (const Elf_Verdaux *)
 		((char *)vd + vd->vd_aux);
 	    if (strcmp(vername, depobj->strtab + aux->vda_name) == 0)
 		return (0);
 	}
 	if (vd->vd_next == 0)
 	    break;
 	vd = (const Elf_Verdef *) ((char *)vd + vd->vd_next);
     }
     if (vna->vna_flags & VER_FLG_WEAK)
 	return (0);
     _rtld_error("%s: version %s required by %s not found",
 	depobj->path, vername, refobj->path);
     return (-1);
 }
 
 static int
 rtld_verify_object_versions(Obj_Entry *obj)
 {
     const Elf_Verneed *vn;
     const Elf_Verdef  *vd;
     const Elf_Verdaux *vda;
     const Elf_Vernaux *vna;
     const Obj_Entry *depobj;
     int maxvernum, vernum;
 
     if (obj->ver_checked)
 	return (0);
     obj->ver_checked = true;
 
     maxvernum = 0;
     /*
      * Walk over defined and required version records and figure out
      * max index used by any of them. Do very basic sanity checking
      * while there.
      */
     vn = obj->verneed;
     while (vn != NULL) {
 	if (vn->vn_version != VER_NEED_CURRENT) {
 	    _rtld_error("%s: Unsupported version %d of Elf_Verneed entry",
 		obj->path, vn->vn_version);
 	    return (-1);
 	}
 	vna = (const Elf_Vernaux *) ((char *)vn + vn->vn_aux);
 	for (;;) {
 	    vernum = VER_NEED_IDX(vna->vna_other);
 	    if (vernum > maxvernum)
 		maxvernum = vernum;
 	    if (vna->vna_next == 0)
 		 break;
 	    vna = (const Elf_Vernaux *) ((char *)vna + vna->vna_next);
 	}
 	if (vn->vn_next == 0)
 	    break;
 	vn = (const Elf_Verneed *) ((char *)vn + vn->vn_next);
     }
 
     vd = obj->verdef;
     while (vd != NULL) {
 	if (vd->vd_version != VER_DEF_CURRENT) {
 	    _rtld_error("%s: Unsupported version %d of Elf_Verdef entry",
 		obj->path, vd->vd_version);
 	    return (-1);
 	}
 	vernum = VER_DEF_IDX(vd->vd_ndx);
 	if (vernum > maxvernum)
 		maxvernum = vernum;
 	if (vd->vd_next == 0)
 	    break;
 	vd = (const Elf_Verdef *) ((char *)vd + vd->vd_next);
     }
 
     if (maxvernum == 0)
 	return (0);
 
     /*
      * Store version information in array indexable by version index.
      * Verify that object version requirements are satisfied along the
      * way.
      */
     obj->vernum = maxvernum + 1;
     obj->vertab = xcalloc(obj->vernum, sizeof(Ver_Entry));
 
     vd = obj->verdef;
     while (vd != NULL) {
 	if ((vd->vd_flags & VER_FLG_BASE) == 0) {
 	    vernum = VER_DEF_IDX(vd->vd_ndx);
 	    assert(vernum <= maxvernum);
 	    vda = (const Elf_Verdaux *)((char *)vd + vd->vd_aux);
 	    obj->vertab[vernum].hash = vd->vd_hash;
 	    obj->vertab[vernum].name = obj->strtab + vda->vda_name;
 	    obj->vertab[vernum].file = NULL;
 	    obj->vertab[vernum].flags = 0;
 	}
 	if (vd->vd_next == 0)
 	    break;
 	vd = (const Elf_Verdef *) ((char *)vd + vd->vd_next);
     }
 
     vn = obj->verneed;
     while (vn != NULL) {
 	depobj = locate_dependency(obj, obj->strtab + vn->vn_file);
 	if (depobj == NULL)
 	    return (-1);
 	vna = (const Elf_Vernaux *) ((char *)vn + vn->vn_aux);
 	for (;;) {
 	    if (check_object_provided_version(obj, depobj, vna))
 		return (-1);
 	    vernum = VER_NEED_IDX(vna->vna_other);
 	    assert(vernum <= maxvernum);
 	    obj->vertab[vernum].hash = vna->vna_hash;
 	    obj->vertab[vernum].name = obj->strtab + vna->vna_name;
 	    obj->vertab[vernum].file = obj->strtab + vn->vn_file;
 	    obj->vertab[vernum].flags = (vna->vna_other & VER_NEED_HIDDEN) ?
 		VER_INFO_HIDDEN : 0;
 	    if (vna->vna_next == 0)
 		 break;
 	    vna = (const Elf_Vernaux *) ((char *)vna + vna->vna_next);
 	}
 	if (vn->vn_next == 0)
 	    break;
 	vn = (const Elf_Verneed *) ((char *)vn + vn->vn_next);
     }
     return 0;
 }
 
 static int
 rtld_verify_versions(const Objlist *objlist)
 {
     Objlist_Entry *entry;
     int rc;
 
     rc = 0;
     STAILQ_FOREACH(entry, objlist, link) {
 	/*
 	 * Skip dummy objects or objects that have their version requirements
 	 * already checked.
 	 */
 	if (entry->obj->strtab == NULL || entry->obj->vertab != NULL)
 	    continue;
 	if (rtld_verify_object_versions(entry->obj) == -1) {
 	    rc = -1;
 	    if (ld_tracing == NULL)
 		break;
 	}
     }
     if (rc == 0 || ld_tracing != NULL)
     	rc = rtld_verify_object_versions(&obj_rtld);
     return rc;
 }
 
 const Ver_Entry *
 fetch_ventry(const Obj_Entry *obj, unsigned long symnum)
 {
     Elf_Versym vernum;
 
     if (obj->vertab) {
 	vernum = VER_NDX(obj->versyms[symnum]);
 	if (vernum >= obj->vernum) {
 	    _rtld_error("%s: symbol %s has wrong verneed value %d",
 		obj->path, obj->strtab + symnum, vernum);
 	} else if (obj->vertab[vernum].hash != 0) {
 	    return &obj->vertab[vernum];
 	}
     }
     return NULL;
 }
 
 int
 _rtld_get_stack_prot(void)
 {
 
 	return (stack_prot);
 }
 
 int
 _rtld_is_dlopened(void *arg)
 {
 	Obj_Entry *obj;
 	RtldLockState lockstate;
 	int res;
 
 	rlock_acquire(rtld_bind_lock, &lockstate);
 	obj = dlcheck(arg);
 	if (obj == NULL)
 		obj = obj_from_addr(arg);
 	if (obj == NULL) {
 		_rtld_error("No shared object contains address");
 		lock_release(rtld_bind_lock, &lockstate);
 		return (-1);
 	}
 	res = obj->dlopened ? 1 : 0;
 	lock_release(rtld_bind_lock, &lockstate);
 	return (res);
 }
 
 static void
 map_stacks_exec(RtldLockState *lockstate)
 {
 	void (*thr_map_stacks_exec)(void);
 
 	if ((max_stack_flags & PF_X) == 0 || (stack_prot & PROT_EXEC) != 0)
 		return;
 	thr_map_stacks_exec = (void (*)(void))(uintptr_t)
 	    get_program_var_addr("__pthread_map_stacks_exec", lockstate);
 	if (thr_map_stacks_exec != NULL) {
 		stack_prot |= PROT_EXEC;
 		thr_map_stacks_exec();
 	}
 }
 
 void
 symlook_init(SymLook *dst, const char *name)
 {
 
 	bzero(dst, sizeof(*dst));
 	dst->name = name;
 	dst->hash = elf_hash(name);
 	dst->hash_gnu = gnu_hash(name);
 }
 
 static void
 symlook_init_from_req(SymLook *dst, const SymLook *src)
 {
 
 	dst->name = src->name;
 	dst->hash = src->hash;
 	dst->hash_gnu = src->hash_gnu;
 	dst->ventry = src->ventry;
 	dst->flags = src->flags;
 	dst->defobj_out = NULL;
 	dst->sym_out = NULL;
 	dst->lockstate = src->lockstate;
 }
 
 
 /*
  * Parse a file descriptor number without pulling in more of libc (e.g. atoi).
  */
 static int
 parse_libdir(const char *str)
 {
 	static const int RADIX = 10;  /* XXXJA: possibly support hex? */
 	const char *orig;
 	int fd;
 	char c;
 
 	orig = str;
 	fd = 0;
 	for (c = *str; c != '\0'; c = *++str) {
 		if (c < '0' || c > '9')
 			return (-1);
 
 		fd *= RADIX;
 		fd += c - '0';
 	}
 
 	/* Make sure we actually parsed something. */
 	if (str == orig) {
 		_rtld_error("failed to parse directory FD from '%s'", str);
 		return (-1);
 	}
 	return (fd);
 }
 
 /*
  * Overrides for libc_pic-provided functions.
  */
 
 int
 __getosreldate(void)
 {
 	size_t len;
 	int oid[2];
 	int error, osrel;
 
 	if (osreldate != 0)
 		return (osreldate);
 
 	oid[0] = CTL_KERN;
 	oid[1] = KERN_OSRELDATE;
 	osrel = 0;
 	len = sizeof(osrel);
 	error = sysctl(oid, 2, &osrel, &len, NULL, 0);
 	if (error == 0 && osrel > 0 && len == sizeof(osrel))
 		osreldate = osrel;
 	return (osreldate);
 }
 
 void
 exit(int status)
 {
 
 	_exit(status);
 }
 
 void (*__cleanup)(void);
 int __isthreaded = 0;
 int _thread_autoinit_dummy_decl = 1;
 
 /*
  * No unresolved symbols for rtld.
  */
 void
 __pthread_cxa_finalize(struct dl_phdr_info *a)
 {
 }
 
 void
 __stack_chk_fail(void)
 {
 
 	_rtld_error("stack overflow detected; terminated");
 	die();
 }
 __weak_reference(__stack_chk_fail, __stack_chk_fail_local);
 
 void
 __chk_fail(void)
 {
 
 	_rtld_error("buffer overflow detected; terminated");
 	die();
 }
 
 const char *
 rtld_strerror(int errnum)
 {
 
 	if (errnum < 0 || errnum >= sys_nerr)
 		return ("Unknown error");
 	return (sys_errlist[errnum]);
 }
Index: projects/clang360-import/share/man/man9/contigmalloc.9
===================================================================
--- projects/clang360-import/share/man/man9/contigmalloc.9	(revision 277944)
+++ projects/clang360-import/share/man/man9/contigmalloc.9	(revision 277945)
@@ -1,132 +1,139 @@
 .\"
 .\" Copyright (c) 2004 Joseph Koshy
 .\" All rights reserved.
 .\"
 .\" Redistribution and use in source and binary forms, with or without
 .\" modification, are permitted provided that the following conditions
 .\" are met:
 .\" 1. Redistributions of source code must retain the above copyright
 .\"    notice, this list of conditions and the following disclaimer.
 .\" 2. Redistributions in binary form must reproduce the above copyright
 .\"    notice, this list of conditions and the following disclaimer in the
 .\"    documentation and/or other materials provided with the distribution.
 .\"
 .\" THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS''
 .\" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED
 .\" TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
 .\" PURPOSE ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE
 .\" LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR
 .\" CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF
 .\" SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS
 .\" INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN
 .\" CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
 .\" ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
 .\" POSSIBILITY OF SUCH DAMAGE.
 .\"
 .\" $FreeBSD$
 .\"
-.Dd July 19, 2007
+.Dd January 29, 2015
 .Dt CONTIGMALLOC 9
 .Os
 .Sh NAME
 .Nm contigmalloc , contigfree
 .Nd manage contiguous kernel physical memory
 .Sh SYNOPSIS
 .In sys/types.h
 .In sys/malloc.h
 .Ft "void *"
 .Fo contigmalloc
 .Fa "unsigned long size"
 .Fa "struct malloc_type *type"
 .Fa "int flags"
 .Fa "vm_paddr_t low"
 .Fa "vm_paddr_t high"
 .Fa "unsigned long alignment"
 .Fa "vm_paddr_t boundary"
 .Fc
 .Ft void
 .Fo contigfree
 .Fa "void *addr"
 .Fa "unsigned long size"
 .Fa "struct malloc_type *type"
 .Fc
 .Sh DESCRIPTION
 The
 .Fn contigmalloc
 function allocates
 .Fa size
 bytes of contiguous physical memory that is aligned to
 .Fa alignment
 bytes, and which does not cross a boundary of
 .Fa boundary
 bytes.
 If successful, the allocation will reside between physical addresses
 .Fa low
 and
 .Fa high .
 The returned pointer points to a wired kernel virtual
 address range of
 .Fa size
 bytes allocated from the kernel virtual address (KVA) map.
 .Pp
 The
 .Fa flags
 parameter modifies
 .Fn contigmalloc Ns 's
 behaviour as follows:
 .Bl -tag -width indent
 .It Dv M_ZERO
 Causes the allocated physical memory to be zero filled.
 .It Dv M_NOWAIT
 Causes
 .Fn contigmalloc
 to return
 .Dv NULL
 if the request cannot be immediately fulfilled due to resource shortage.
 .El
 .Pp
 Other flags (if present) are ignored.
 .Pp
 The
 .Fn contigfree
 function deallocates memory allocated by a previous call to
 .Fn contigmalloc .
 .Sh IMPLEMENTATION NOTES
 The
 .Fn contigmalloc
 function does not sleep waiting for memory resources to be freed up,
 but instead actively reclaims pages before giving up.
 However, unless
 .Dv M_NOWAIT
 is specified, it may select a page for reclamation that must first be
 written to backing storage, causing it to sleep.
+.Pp
+The
+.Fn contigfree
+function does not accept
+.Dv NULL
+as an address input, unlike
+.Xr free 9 .
 .Sh RETURN VALUES
 The
 .Fn contigmalloc
 function returns a kernel virtual address if allocation succeeds,
 or
 .Dv NULL
 otherwise.
 .Sh EXAMPLES
 .Bd -literal
 void *p;
 p = contigmalloc(8192, M_DEVBUF, M_ZERO, 0, (1L << 22),
     32 * 1024, 1024 * 1024);
 .Ed
 .Pp
 Ask for 8192 bytes of zero-filled memory residing between physical
 address 0 and 4194303 inclusive, aligned to a 32K boundary and not
 crossing a 1M address boundary.
 .Sh DIAGNOSTICS
 The
 .Fn contigmalloc
 function will panic if
 .Fa size
 is zero, or if
 .Fa alignment
 or
 .Fa boundary
 is not a power of two.
 .Sh SEE ALSO
 .Xr malloc 9 ,
 .Xr memguard 9
Index: projects/clang360-import/share/misc/committers-src.dot
===================================================================
--- projects/clang360-import/share/misc/committers-src.dot	(revision 277944)
+++ projects/clang360-import/share/misc/committers-src.dot	(revision 277945)
@@ -1,713 +1,717 @@
 # $FreeBSD$
 
 # This file is meant to list all FreeBSD src committers and describe the
 # mentor-mentee relationships between them.
 # The graphical output can be generated from this file with the following
 # command:
 # $ dot -T png -o file.png committers-src.dot
 #
 # The dot binary is part of the graphics/graphviz port.
 
 digraph src {
 
 # Node definitions follow this example:
 #
 #   foo [label="Foo Bar\nfoo@FreeBSD.org\n????/??/??"]
 #
 # ????/??/?? is the date when the commit bit was obtained, usually the one you
 # can find looking at svn logs for the svnadmin/access file.
 # Use YYYY/MM/DD format.
 #
 # For returned commit bits, the node definition will follow this example:
 #
 #   foo [label="Foo Bar\nfoo@FreeBSD.org\n????/??/??\n????/??/??"]
 #
 # The first date is the same as for an active committer, the second date is
 # the date when the commit bit has been returned. Again, check svn logs.
 
 node [color=grey62, style=filled, bgcolor=black];
 
 # Alumni go here.. Try to keep things sorted.
 
 alm [label="Andrew Moore\nalm@FreeBSD.org\n1993/06/12\n????/??/??"]
 anholt [label="Eric Anholt\nanholt@FreeBSD.org\n2002/04/22\n2008/08/07"]
 archie [label="Archie Cobbs\narchie@FreeBSD.org\n1998/11/06\n2006/06/09"]
 arr [label="Andrew R. Reiter\narr@FreeBSD.org\n2001/11/02\n2005/05/25"]
 arun [label="Arun Sharma\narun@FreeBSD.org\n2003/03/06\n2006/12/16"]
 asmodai [label="Jeroen Ruigrok\nasmodai@FreeBSD.org\n1999/12/16\n2001/11/16"]
 benjsc [label="Benjamin Close\nbenjsc@FreeBSD.org\n2007/02/09\n2010/09/15"]
 billf [label="Bill Fumerola\nbillf@FreeBSD.org\n1998/11/11\n2008/11/10"]
 bmah [label="Bruce A. Mah\nbmah@FreeBSD.org\n2002/01/29\n2009/09/13"]
 bmilekic [label="Bosko Milekic\nbmilekic@FreeBSD.org\n2000/09/21\n2008/11/10"]
 bushman [label="Michael Bushkov\nbushman@FreeBSD.org\n2007/03/10\n2010/04/29"]
 carl [label="Carl Delsey\ncarl@FreeBSD.org\n2013/01/14\n2014/03/06"]
 ceri [label="Ceri Davies\nceri@FreeBSD.org\n2006/11/07\n2012/03/07"]
 cjc [label="Crist J. Clark\ncjc@FreeBSD.org\n2001/06/01\n2006/12/29"]
 davidxu [label="David Xu\ndavidxu@FreeBSD.org\n2002/09/02\n2014/04/14"]
 dds [label="Diomidis Spinellis\ndds@FreeBSD.org\n2003/06/20\n2010/09/22"]
 dhartmei [label="Daniel Hartmeier\ndhartmei@FreeBSD.org\n2004/04/06\n2008/12/08"]
 dmlb [label="Duncan Barclay\ndmlb@FreeBSD.org\n2001/12/14\n2008/11/10"]
 dougb [label="Doug Barton\ndougb@FreeBSD.org\n2000/10/26\n2012/10/08"]
 eik [label="Oliver Eikemeier\neik@FreeBSD.org\n2004/05/20\n2008/11/10"]
 furuta [label="Atsushi Furuta\nfuruta@FreeBSD.org\n2000/06/21\n2003/03/08"]
 gj [label="Gary L. Jennejohn\ngj@FreeBSD.org\n1994/??/??\n2006/04/28"]
 groudier [label="Gerard Roudier\ngroudier@FreeBSD.org\n1999/12/30\n2006/04/06"]
 jake [label="Jake Burkholder\njake@FreeBSD.org\n2000/05/16\n2008/11/10"]
 jayanth [label="Jayanth Vijayaraghavan\njayanth@FreeBSD.org\n2000/05/08\n2008/11/10"]
 jb [label="John Birrell\njb@FreeBSD.org\n1997/03/27\n2009/12/15"]
 jdp [label="John Polstra\njdp@FreeBSD.org\n1995/12/07\n2008/02/26"]
 jedgar [label="Chris D. Faulhaber\njedgar@FreeBSD.org\n1999/12/15\n2006/04/07"]
 jkh [label="Jordan K. Hubbard\njkh@FreeBSD.org\n1993/06/12\n2008/06/13"]
 jlemon [label="Jonathan Lemon\njlemon@FreeBSD.org\n1997/08/14\n2008/11/10"]
 joe [label="Josef Karthauser\njoe@FreeBSD.org\n1999/10/22\n2008/08/10"]
 jtc [label="J.T. Conklin\njtc@FreeBSD.org\n1993/06/12\n????/??/??"]
 kbyanc [label="Kelly Yancey\nkbyanc@FreeBSD.org\n2000/07/11\n2006/07/25"]
 keichii [label="Michael Wu\nkeichii@FreeBSD.org\n2001/03/07\n2006/04/28"]
 linimon [label="Mark Linimon\nlinimon@FreeBSD.org\n2006/09/30\n2008/05/04"]
 lulf [label="Ulf Lilleengen\nlulf@FreeBSD.org\n2007/10/24\n2012/01/19"]
 mb [label="Maxim Bolotin\nmb@FreeBSD.org\n2000/04/06\n2003/03/08"]
 marks [label="Mark Santcroos\nmarks@FreeBSD.org\n2004/03/18\n2008/09/29"]
 mike [label="Mike Barcroft\nmike@FreeBSD.org\n2001/07/17\n2006/04/28"]
 msmith [label="Mike Smith\nmsmith@FreeBSD.org\n1996/10/22\n2003/12/15"]
 murray [label="Murray Stokely\nmurray@FreeBSD.org\n2000/04/05\n2010/07/25"]
 mux [label="Maxime Henrion\nmux@FreeBSD.org\n2002/03/03\n2011/06/22"]
 nate [label="Nate Willams\nnate@FreeBSD.org\n1993/06/12\n2003/12/15"]
 njl [label="Nate Lawson\nnjl@FreeBSD.org\n2002/08/07\n2008/02/16"]
 non [label="Noriaki Mitsnaga\nnon@FreeBSD.org\n2000/06/19\n2007/03/06"]
 onoe [label="Atsushi Onoe\nonoe@FreeBSD.org\n2000/07/21\n2008/11/10"]
 rafan [label="Rong-En Fan\nrafan@FreeBSD.org\n2007/01/31\n2012/07/23"]
 randi [label="Randi Harper\nrandi@FreeBSD.org\n2010/04/20\n2012/05/10"]
 rgrimes [label="Rod Grimes\nrgrimes@FreeBSD.org\n1993/06/12\n2003/03/08"]
 rink [label="Rink Springer\nrink@FreeBSD.org\n2006/01/16\n2010/11/04"]
 robert [label="Robert Drehmel\nrobert@FreeBSD.org\n2001/08/23\n2006/05/13"]
 sah [label="Sam Hopkins\nsah@FreeBSD.org\n2004/12/15\n2008/11/10"]
 shafeeq [label="Shafeeq Sinnamohideen\nshafeeq@FreeBSD.org\n2000/06/19\n2006/04/06"]
 sheldonh [label="Sheldon Hearn\nsheldonh@FreeBSD.org\n1999/06/14\n2006/05/13"]
 shiba [label="Takeshi Shibagaki\nshiba@FreeBSD.org\n2000/06/19\n2008/11/10"]
 shin [label="Yoshinobu Inoue\nshin@FreeBSD.org\n1999/07/29\n2003/03/08"]
 snb [label="Nick Barkas\nsnb@FreeBSD.org\n2009/05/05\n2010/11/04"]
 tmm [label="Thomas Moestl\ntmm@FreeBSD.org\n2001/03/07\n2006/07/12"]
 toshi [label="Toshihiko Arai\ntoshi@FreeBSD.org\n2000/07/06\n2003/03/08"]
 tshiozak [label="Takuya SHIOZAKI\ntshiozak@FreeBSD.org\n2001/04/25\n2003/03/08"]
 uch [label="UCHIYAMA Yasushi\nuch@FreeBSD.org\n2000/06/21\n2002/04/24"]
 wilko [label="Wilko Bulte\nwilko@FreeBSD.org\n2000/01/13\n2013/01/17"]
 yar [label="Yar Tikhiy\nyar@FreeBSD.org\n2001/03/25\n2012/05/23"]
 zack [label="Zack Kirsch\nzack@FreeBSD.org\n2010/11/05\n2012/09/08"]
 
 
 node [color=lightblue2, style=filled, bgcolor=black];
 
 # Current src committers go here. Try to keep things sorted.
 
 ache [label="Andrey Chernov\nache@FreeBSD.org\n1993/10/31"]
 achim [label="Achim Leubner\nachim@FreeBSD.org\n2013/01/23"]
 adrian [label="Adrian Chadd\nadrian@FreeBSD.org\n2000/07/03"]
 ae [label="Andrey V. Elsukov\nae@FreeBSD.org\n2010/06/03"]
 akiyama [label="Shunsuke Akiyama\nakiyama@FreeBSD.org\n2000/06/19"]
 alc [label="Alan Cox\nalc@FreeBSD.org\n1999/02/23"]
 ambrisko [label="Doug Ambrisko\nambrisko@FreeBSD.org\n2001/12/19"]
 anchie [label="Ana Kukec\nanchie@FreeBSD.org\n2010/04/14"]
 andre [label="Andre Oppermann\nandre@FreeBSD.org\n2003/11/12"]
 andreast [label="Andreas Tobler\nandreast@FreeBSD.org\n2010/09/05"]
 andrew [label="Andrew Turner\nandrew@FreeBSD.org\n2010/07/19"]
 antoine [label="Antoine Brodin\nantoine@FreeBSD.org\n2008/02/03"]
 ariff [label="Ariff Abdullah\nariff@FreeBSD.org\n2005/11/14"]
 art [label="Artem Belevich\nart@FreeBSD.org\n2011/03/29"]
 arybchik [label="Andrew Rybchenko\narybchik@FreeBSD.org\n2014/10/12"]
 asomers [label="Alan Somers\nasomers@FreeBSD.org\n2013/04/24"]
 avg [label="Andriy Gapon\navg@FreeBSD.org\n2009/02/18"]
 bapt [label="Baptiste Daroussin\nbapt@FreeBSD.org\n2011/12/23"]
 bdrewery [label="Bryan Drewery\nbdrewery@FreeBSD.org\n2013/12/14"]
 benl [label="Ben Laurie\nbenl@FreeBSD.org\n2011/05/18"]
 benno [label="Benno Rice\nbenno@FreeBSD.org\n2000/11/02"]
 bms [label="Bruce M Simpson\nbms@FreeBSD.org\n2003/08/06"]
 br [label="Ruslan Bukin\nbr@FreeBSD.org\n2013/09/02"]
 brian [label="Brian Somers\nbrian@FreeBSD.org\n1996/12/16"]
 brooks [label="Brooks Davis\nbrooks@FreeBSD.org\n2001/06/21"]
 brucec [label="Bruce Cran\nbrucec@FreeBSD.org\n2010/01/29"]
 brueffer [label="Christian Brueffer\nbrueffer@FreeBSD.org\n2006/02/28"]
 bruno [label="Bruno Ducrot\nbruno@FreeBSD.org\n2005/07/18"]
 bryanv [label="Bryan Venteicher\nbryanv@FreeBSD.org\n2012/11/03"]
 bschmidt [label="Bernhard Schmidt\nbschmidt@FreeBSD.org\n2010/02/06"]
 bz [label="Bjoern A. Zeeb\nbz@FreeBSD.org\n2004/07/27"]
 cognet [label="Olivier Houchard\ncognet@FreeBSD.org\n2002/10/09"]
 cokane [label="Coleman Kane\ncokane@FreeBSD.org\n2000/06/19"]
 cperciva [label="Colin Percival\ncperciva@FreeBSD.org\n2004/01/20"]
 csjp [label="Christian S.J. Peron\ncsjp@FreeBSD.org\n2004/05/04"]
 das [label="David Schultz\ndas@FreeBSD.org\n2003/02/21"]
 davide [label="Davide Italiano\ndavide@FreeBSD.org\n2012/01/27"]
 dchagin [label="Dmitry Chagin\ndchagin@FreeBSD.org\n2009/02/28"]
 delphij [label="Xin Li\ndelphij@FreeBSD.org\n2004/09/14"]
 des [label="Dag-Erling Smorgrav\ndes@FreeBSD.org\n1998/04/03"]
 dfr [label="Doug Rabson\ndfr@FreeBSD.org\n????/??/??"]
 dg [label="David Greenman\ndg@FreeBSD.org\n1993/06/14"]
 dim [label="Dimitry Andric\ndim@FreeBSD.org\n2010/08/30"]
 dteske [label="Devin Teske\ndteske@FreeBSD.org\n2012/04/10"]
 dumbbell [label="Jean-Sebastien Pedron\ndumbbell@FreeBSD.org\n2004/11/29"]
 dwmalone [label="David Malone\ndwmalone@FreeBSD.org\n2000/07/11"]
 eadler [label="Eitan Adler\neadler@FreeBSD.org\n2012/01/18"]
 ed [label="Ed Schouten\ned@FreeBSD.org\n2008/05/22"]
 edavis [label="Eric Davis\nedavis@FreeBSD.org\n2013/10/09"]
 edwin [label="Edwin Groothuis\nedwin@FreeBSD.org\n2007/06/25"]
 eivind [label="Eivind Eklund\neivind@FreeBSD.org\n1997/02/02"]
 emaste [label="Ed Maste\nemaste@FreeBSD.org\n2005/10/04"]
 emax [label="Maksim Yevmenkin\nemax@FreeBSD.org\n2003/10/12"]
 eri [label="Ermal Luci\neri@FreeBSD.org\n2008/06/11"]
+erj [label="Eric Joyner\nerj@FreeBSD.org\n2014/12/14"]
 fabient [label="Fabien Thomas\nfabient@FreeBSD.org\n2009/03/16"]
 fanf [label="Tony Finch\nfanf@FreeBSD.org\n2002/05/05"]
 fjoe [label="Max Khon\nfjoe@FreeBSD.org\n2001/08/06"]
 flz [label="Florent Thoumie\nflz@FreeBSD.org\n2006/03/30"]
 gabor [label="Gabor Kovesdan\ngabor@FreeBSD.org\n2010/02/02"]
 gad [label="Garance A. Drosehn\ngad@FreeBSD.org\n2000/10/27"]
 gallatin [label="Andrew Gallatin\ngallatin@FreeBSD.org\n1999/01/15"]
 gavin [label="Gavin Atkinson\ngavin@FreeBSD.org\n2009/12/07"]
 gibbs [label="Justin T. Gibbs\ngibbs@FreeBSD.org\n????/??/??"]
 gjb [label="Glen Barber\ngjb@FreeBSD.org\n2013/06/04"]
 gleb [label="Gleb Kurtsou\ngleb@FreeBSD.org\n2011/09/19"]
 glebius [label="Gleb Smirnoff\nglebius@FreeBSD.org\n2004/07/14"]
 gnn [label="George V. Neville-Neil\ngnn@FreeBSD.org\n2004/10/11"]
 gordon [label="Gordon Tetlow\ngordon@FreeBSD.org\n2002/05/17"]
 grehan [label="Peter Grehan\ngrehan@FreeBSD.org\n2002/08/08"]
 grog [label="Greg Lehey\ngrog@FreeBSD.org\n1998/08/30"]
 gshapiro [label="Gregory Shapiro\ngshapiro@FreeBSD.org\n2000/07/12"]
 harti [label="Hartmut Brandt\nharti@FreeBSD.org\n2003/01/29"]
 hiren [label="Hiren Panchasara\nhiren@FreeBSD.org\n2013/04/12"]
 hmp [label="Hiten Pandya\nhmp@FreeBSD.org\n2004/03/23"]
 ian [label="Ian Lepore\nian@FreeBSD.org\n2013/01/07"]
 iedowse [label="Ian Dowse\niedowse@FreeBSD.org\n2000/12/01"]
 imp [label="Warner Losh\nimp@FreeBSD.org\n1996/09/20"]
 ivoras [label="Ivan Voras\nivoras@FreeBSD.org\n2008/06/10"]
 jamie [label="Jamie Gritton\njamie@FreeBSD.org\n2009/01/28"]
 jasone [label="Jason Evans\njasone@FreeBSD.org\n1999/03/03"]
 jceel [label="Jakub Klama\njceel@FreeBSD.org\n2011/09/25"]
 jch [label="Julien Charbon\njch@FreeBSD.org\n2014/09/24"]
 jchandra [label="Jayachandran C.\njchandra@FreeBSD.org\n2010/05/19"]
 jeff [label="Jeff Roberson\njeff@FreeBSD.org\n2002/02/21"]
 jh [label="Jaakko Heinonen\njh@FreeBSD.org\n2009/10/02"]
 jhb [label="John Baldwin\njhb@FreeBSD.org\n1999/08/23"]
 jhibbits [label="Justin Hibbits\njhibbits@FreeBSD.org\n2011/11/30"]
 jilles [label="Jilles Tjoelker\njilles@FreeBSD.org\n2009/05/22"]
 jimharris [label="Jim Harris\njimharris@FreeBSD.org\n2011/12/09"]
 jinmei [label="JINMEI Tatuya\njinmei@FreeBSD.org\n2007/03/17"]
 jkim [label="Jung-uk Kim\njkim@FreeBSD.org\n2005/07/06"]
 jkoshy [label="A. Joseph Koshy\njkoshy@FreeBSD.org\n1998/05/13"]
 jlh [label="Jeremie Le Hen\njlh@FreeBSD.org\n2012/04/22"]
 jls [label="Jordan Sissel\njls@FreeBSD.org\n2006/12/06"]
 jmg [label="John-Mark Gurney\njmg@FreeBSD.org\n1997/02/13"]
 jmmv [label="Julio Merino\njmmv@FreeBSD.org\n2013/11/02"]
 joerg [label="Joerg Wunsch\njoerg@FreeBSD.org\n1993/11/14"]
 jon [label="Jonathan Chen\njon@FreeBSD.org\n2000/10/17"]
 jonathan [label="Jonathan Anderson\njonathan@FreeBSD.org\n2010/10/07"]
 jpaetzel [label="Josh Paetzel\njpaetzel@FreeBSD.org\n2011/01/21"]
 julian [label="Julian Elischer\njulian@FreeBSD.org\n1993/04/19"]
 jwd [label="John De Boskey\njwd@FreeBSD.org\n2000/05/19"]
 kaiw [label="Kai Wang\nkaiw@FreeBSD.org\n2007/09/26"]
 kan [label="Alexander Kabaev\nkan@FreeBSD.org\n2002/07/21"]
 kargl [label="Steven G. Kargl\nkargl@FreeBSD.org\n2011/01/17"]
 ken [label="Ken Merry\nken@FreeBSD.org\n1998/09/08"]
 kensmith [label="Ken Smith\nkensmith@FreeBSD.org\n2004/01/23"]
 kevlo [label="Kevin Lo\nkevlo@FreeBSD.org\n2006/07/23"]
 kib [label="Konstantin Belousov\nkib@FreeBSD.org\n2006/06/03"]
 kmacy [label="Kip Macy\nkmacy@FreeBSD.org\n2005/06/01"]
 le [label="Lukas Ertl\nle@FreeBSD.org\n2004/02/02"]
 loos [label="Luiz Otavio O Souza\nloos@FreeBSD.org\n2013/07/03"]
 lstewart [label="Lawrence Stewart\nlstewart@FreeBSD.org\n2008/10/06"]
 marcel [label="Marcel Moolenaar\nmarcel@FreeBSD.org\n1999/07/03"]
 marius [label="Marius Strobl\nmarius@FreeBSD.org\n2004/04/17"]
 markj [label="Mark Johnston\nmarkj@FreeBSD.org\n2012/12/18"]
 markm [label="Mark Murray\nmarkm@FreeBSD.org\n1995/04/24"]
 markus [label="Markus Brueffer\nmarkus@FreeBSD.org\n2006/06/01"]
 matteo [label="Matteo Riondato\nmatteo@FreeBSD.org\n2006/01/18"]
 mav [label="Alexander Motin\nmav@FreeBSD.org\n2007/04/12"]
 maxim [label="Maxim Konovalov\nmaxim@FreeBSD.org\n2002/02/07"]
 mdf [label="Matthew Fleming\nmdf@FreeBSD.org\n2010/06/04"]
 mdodd [label="Matthew N. Dodd\nmdodd@FreeBSD.org\n1999/07/27"]
 melifaro [label="Alexander V. Chernikov\nmelifaro@FreeBSD.org\n2011/10/04"]
 mjacob [label="Matt Jacob\nmjacob@FreeBSD.org\n1997/08/13"]
 mjg [label="Mateusz Guzik\nmjg@FreeBSD.org\n2012/06/04"]
 mlaier [label="Max Laier\nmlaier@FreeBSD.org\n2004/02/10"]
 monthadar [label="Monthadar Al Jaberi\nmonthadar@FreeBSD.org\n2012/04/02"]
 mp [label="Mark Peek\nmp@FreeBSD.org\n2001/07/27"]
 mr [label="Michael Reifenberger\nmr@FreeBSD.org\n2001/09/30"]
 neel [label="Neel Natu\nneel@FreeBSD.org\n2009/09/20"]
 netchild [label="Alexander Leidinger\nnetchild@FreeBSD.org\n2005/03/31"]
 ngie [label="Garrett Cooper\nngie@FreeBSD.org\n2014/07/27"]
 nork [label="Norikatsu Shigemura\nnork@FreeBSD.org\n2009/06/09"]
 np [label="Navdeep Parhar\nnp@FreeBSD.org\n2009/06/05"]
 nwhitehorn [label="Nathan Whitehorn\nnwhitehorn@FreeBSD.org\n2008/07/03"]
 obrien [label="David E. O'Brien\nobrien@FreeBSD.org\n1996/10/29"]
 olli [label="Oliver Fromme\nolli@FreeBSD.org\n2008/02/14"]
 peadar [label="Peter Edwards\npeadar@FreeBSD.org\n2004/03/08"]
 peter [label="Peter Wemm\npeter@FreeBSD.org\n1995/07/04"]
 peterj [label="Peter Jeremy\npeterj@FreeBSD.org\n2012/09/14"]
 pfg [label="Pedro Giffuni\npfg@FreeBSD.org\n2011/12/01"]
 philip [label="Philip Paeps\nphilip@FreeBSD.org\n2004/01/21"]
 phk [label="Poul-Henning Kamp\nphk@FreeBSD.org\n1994/02/21"]
 pho [label="Peter Holm\npho@FreeBSD.org\n2008/11/16"]
 pjd [label="Pawel Jakub Dawidek\npjd@FreeBSD.org\n2004/02/02"]
 pkelsey [label="Patrick Kelsey\pkelsey@FreeBSD.org\n2014/05/29"]
 pluknet [label="Sergey Kandaurov\npluknet@FreeBSD.org\n2010/10/05"]
 ps [label="Paul Saab\nps@FreeBSD.org\n2000/02/23"]
 qingli [label="Qing Li\nqingli@FreeBSD.org\n2005/04/13"]
 ray [label="Aleksandr Rybalko\nray@FreeBSD.org\n2011/05/25"]
 rdivacky [label="Roman Divacky\nrdivacky@FreeBSD.org\n2008/03/13"]
 remko [label="Remko Lodder\nremko@FreeBSD.org\n2007/02/23"]
 rik [label="Roman Kurakin\nrik@FreeBSD.org\n2003/12/18"]
 rmacklem [label="Rick Macklem\nrmacklem@FreeBSD.org\n2009/03/27"]
 rmh [label="Robert Millan\nrmh@FreeBSD.org\n2011/09/18"]
 rnoland [label="Robert Noland\nrnoland@FreeBSD.org\n2008/09/15"]
 roberto [label="Ollivier Robert\nroberto@FreeBSD.org\n1995/02/22"]
 royger [label="Roger Pau Monne\nroyger@FreeBSD.org\n2013/11/26"]
 rpaulo [label="Rui Paulo\nrpaulo@FreeBSD.org\n2007/09/25"]
 rrs [label="Randall R Stewart\nrrs@FreeBSD.org\n2007/02/08"]
 rse [label="Ralf S. Engelschall\nrse@FreeBSD.org\n1997/07/31"]
 rstone [label="Ryan Stone\nrstone@FreeBSD.org\n2010/04/19"]
 ru [label="Ruslan Ermilov\nru@FreeBSD.org\n1999/05/27"]
 rwatson [label="Robert N. M. Watson\nrwatson@FreeBSD.org\n1999/12/16"]
 sam [label="Sam Leffler\nsam@FreeBSD.org\n2002/07/02"]
 sanpei [label="MIHIRA Sanpei Yoshiro\nsanpei@FreeBSD.org\n2000/06/19"]
 sbruno [label="Sean Bruno\nsbruno@FreeBSD.org\n2008/08/02"]
 scf [label="Sean C. Farley\nscf@FreeBSD.org\n2007/06/24"]
 schweikh [label="Jens Schweikhardt\nschweikh@FreeBSD.org\n2001/04/06"]
 scottl [label="Scott Long\nscottl@FreeBSD.org\n2000/09/28"]
 se [label="Stefan Esser\nse@FreeBSD.org\n1994/08/26"]
 sephe [label="Sepherosa Ziehau\nsephe@FreeBSD.org\n2007/03/28"]
 sepotvin [label="Stephane E. Potvin\nsepotvin@FreeBSD.org\n2007/02/15"]
 simon [label="Simon L. Nielsen\nsimon@FreeBSD.org\n2006/03/07"]
 sjg [label="Simon J. Gerraty\nsjg@FreeBSD.org\n2012/10/23"]
 slm [label="Stephen McConnell\nslm@FreeBSD.org\n2014/05/07"]
 smh [label="Steven Hartland\nsmh@FreeBSD.org\n2012/11/12"]
 sobomax [label="Maxim Sobolev\nsobomax@FreeBSD.org\n2001/07/25"]
 sos [label="Soren Schmidt\nsos@FreeBSD.org\n????/??/??"]
 sson [label="Stacey Son\nsson@FreeBSD.org\n2008/07/08"]
 stas [label="Stanislav Sedov\nstas@FreeBSD.org\n2008/08/22"]
 suz [label="SUZUKI Shinsuke\nsuz@FreeBSD.org\n2002/03/26"]
 syrinx [label="Shteryana Shopova\nsyrinx@FreeBSD.org\n2006/10/07"]
 takawata [label="Takanori Watanabe\ntakawata@FreeBSD.org\n2000/07/06"]
 theraven [label="David Chisnall\ntheraven@FreeBSD.org\n2011/11/11"]
 thompsa [label="Andrew Thompson\nthompsa@FreeBSD.org\n2005/05/25"]
 ticso [label="Bernd Walter\nticso@FreeBSD.org\n2002/01/31"]
 tijl [label="Tijl Coosemans\ntijl@FreeBSD.org\n2010/07/16"]
 trasz [label="Edward Tomasz Napierala\ntrasz@FreeBSD.org\n2008/08/22"]
 trhodes [label="Tom Rhodes\ntrhodes@FreeBSD.org\n2002/05/28"]
 trociny [label="Mikolaj Golub\ntrociny@FreeBSD.org\n2011/03/10"]
 tuexen [label="Michael Tuexen\ntuexen@FreeBSD.org\n2009/06/06"]
 tychon [label="Tycho Nightingale\ntychon@FreeBSD.org\n2014/01/21"]
 ume [label="Hajimu UMEMOTO\nume@FreeBSD.org\n2000/02/26"]
 uqs [label="Ulrich Spoerlein\nuqs@FreeBSD.org\n2010/01/28"]
 vanhu [label="Yvan Vanhullebus\nvanhu@FreeBSD.org\n2008/07/21"]
 versus [label="Konrad Jankowski\nversus@FreeBSD.org\n2008/10/27"]
 weongyo [label="Weongyo Jeong\nweongyo@FreeBSD.org\n2007/12/21"]
 wes [label="Wes Peters\nwes@FreeBSD.org\n1998/11/25"]
 wkoszek [label="Wojciech A. Koszek\nwkoszek@FreeBSD.org\n2006/02/21"]
 wollman [label="Garrett Wollman\nwollman@FreeBSD.org\n????/??/??"]
 wsalamon [label="Wayne Salamon\nwsalamon@FreeBSD.org\n2005/06/25"]
 yongari [label="Pyun YongHyeon\nyongari@FreeBSD.org\n2004/08/01"]
 zbb [label="Zbigniew Bodek\nzbb@FreeBSD.org\n2013/09/02"]
 zec [label="Marko Zec\nzec@FreeBSD.org\n2008/06/22"]
 zml [label="Zachary Loafman\nzml@FreeBSD.org\n2009/05/27"]
 zont [label="Andrey Zonov\nzont@FreeBSD.org\n2012/08/21"]
 
 # Pseudo target representing rev 1.1 of commit.allow
 day1 [label="Birth of FreeBSD"]
 
 # Here are the mentor/mentee relationships.
 # Group together all the mentees for a particular mentor.
 # Keep the list sorted by mentor login.
 
 day1 -> jtc
 day1 -> jkh
 day1 -> nate
 day1 -> rgrimes
 day1 -> alm
 day1 -> dg
 
 adrian -> loos
 adrian -> monthadar
 adrian -> ray
 adrian -> rmh
 
 ae -> melifaro
 
 alc -> davide
 
 andre -> qingli
 
 anholt -> jkim
 
 avg -> art
 avg -> pluknet
 avg -> smh
 
 bapt -> bdrewery
 
 benno -> grehan
 
 billf -> dougb
 billf -> gad
 billf -> jedgar
 billf -> jhb
 billf -> shafeeq
 
 bmilekic -> csjp
 
 bms -> dhartmei
 bms -> mlaier
 bms -> thompsa
 
 brian -> joe
 
 brooks -> bushman
 brooks -> jamie
 brooks -> theraven
 
 bz -> anchie
 bz -> jamie
 bz -> syrinx
 
 cognet -> br
 cognet -> jceel
 cognet -> kevlo
 cognet -> ian
 cognet -> wkoszek
 cognet -> zbb
 
 cperciva -> eadler
 cperciva -> flz
 cperciva -> randi
 cperciva -> simon
 
 csjp -> bushman
 
 das -> kargl
 
 delphij -> gabor
 delphij -> rafan
 
 des -> anholt
 des -> hmp
 des -> mike
 des -> olli
 des -> ru
 des -> bapt
 
 dds -> versus
 
 dfr -> gallatin
 dfr -> zml
 
 dg -> peter
 
 dim -> theraven
 
 dwmalone -> fanf
 dwmalone -> peadar
 dwmalone -> snb
 
 ed -> dim
 ed -> gavin
 ed -> jilles
 ed -> rdivacky
 ed -> uqs
 
 eivind -> des
 eivind -> rwatson
 
 emaste -> achim
 emaste -> rstone
 emaste -> dteske
 emaste -> markj
 
 emax -> markus
 
 fjoe -> versus
 
 gallatin -> ticso
 
 gavin -> versus
 
 gibbs -> mjacob
 gibbs -> njl
 gibbs -> royger
 
 glebius -> mav
 
 gnn -> jinmei
 gnn -> rrs
 gnn -> ivoras
 gnn -> vanhu
 gnn -> lstewart
 gnn -> np
 gnn -> davide
 gnn -> arybchik
+gnn -> erj
 
 grehan -> bryanv
 
 grog -> edwin
 grog -> le
 grog -> peterj
 
 imp -> akiyama
 imp -> ambrisko
 imp -> andrew
 imp -> bmah
 imp -> bruno
 imp -> dmlb
 imp -> emax
 imp -> furuta
 imp -> joe
 imp -> jon
 imp -> keichii
 imp -> mb
 imp -> mr
 imp -> neel
 imp -> non
 imp -> nork
 imp -> onoe
 imp -> remko
 imp -> rik
 imp -> rink
 imp -> sanpei
 imp -> shiba
 imp -> takawata
 imp -> toshi
 imp -> uch
 
 jake -> bms
 jake -> gordon
 jake -> harti
 jake -> jeff
 jake -> kmacy
 jake -> robert
 jake -> yongari
 
 jb -> sson
 
 jdp -> fjoe
+
+jfv -> erj
 
 jhb -> arr
 jhb -> avg
 jhb -> jch
 jhb -> jeff
 jhb -> kbyanc
 jhb -> peterj
 jhb -> pfg
 jhb -> rnoland
 
 jimharris -> carl
 
 jkh -> dfr
 jkh -> gj
 jkh -> grog
 jkh -> imp
 jkh -> jlemon
 jkh -> joerg
 jkh -> jwd
 jkh -> msmith
 jkh -> murray
 jkh -> phk
 jkh -> wes
 jkh -> yar
 
 jkoshy -> kaiw
 jkoshy -> fabient
 jkoshy -> rstone
 
 jlemon -> bmilekic
 jlemon -> brooks
 
 jmallett -> pkelsey
 
 jmmv -> ngie
 
 joerg -> brian
 joerg -> eik
 joerg -> jmg
 joerg -> le
 joerg -> netchild
 joerg -> schweikh
 
 julian -> glebius
 julian -> davidxu
 julian -> archie
 julian -> adrian
 julian -> zec
 julian -> mp
 
 kan -> kib
 
 ken -> asomers
 ken -> slm
 
 kib -> ae
 kib -> dchagin
 kib -> gjb
 kib -> jlh
 kib -> jpaetzel
 kib -> lulf
 kib -> melifaro
 kib -> pho
 kib -> pluknet
 kib -> rdivacky
 kib -> rmacklem
 kib -> rmh
 kib -> stas
 kib -> tijl
 kib -> trociny
 kib -> zont
 
 kmacy -> lstewart
 
 marcel -> art
 marcel -> arun
 marcel -> marius
 marcel -> nwhitehorn
 marcel -> sjg
 
 markm -> jasone
 markm -> sheldonh
 
 mav -> ae
 
 mdf -> gleb
 
 mdodd -> jake
 
 mike -> das
 
 mlaier -> benjsc
 mlaier -> dhartmei
 mlaier -> thompsa
 mlaier -> eri
 
 msmith -> cokane
 msmith -> jasone
 msmith -> scottl
 
 murray -> delphij
 
 mux -> cognet
 mux -> dumbbell
 
 netchild -> ariff
 
 njl -> marks
 njl -> philip
 njl -> rpaulo
 njl -> sepotvin
 
 nwhitehorn -> andreast
 nwhitehorn -> jhibbits
 
 obrien -> benno
 obrien -> groudier
 obrien -> gshapiro
 obrien -> kan
 obrien -> sam
 
 peter -> asmodai
 peter -> jayanth
 peter -> ps
 
 philip -> benl
 philip -> ed
 philip -> jls
 philip -> matteo
 philip -> uqs
 
 phk -> jkoshy
 phk -> mux
 
 pjd -> kib
 pjd -> lulf
 pjd -> smh
 pjd -> trociny
 
 rgrimes -> markm
 
 rmacklem -> jwd
 
 rpaulo -> avg
 rpaulo -> bschmidt
 rpaulo -> dim
 rpaulo -> jmmv
 rpaulo -> ngie
 
 rrs -> brucec
 rrs -> jchandra
 rrs -> tuexen
 
 rstone -> markj
 
 ru -> ceri
 ru -> cjc
 ru -> eik
 ru -> maxim
 ru -> sobomax
 
 rwatson -> adrian
 rwatson -> antoine
 rwatson -> bmah
 rwatson -> brueffer
 rwatson -> bz
 rwatson -> cperciva
 rwatson -> emaste
 rwatson -> gnn
 rwatson -> jh
 rwatson -> jonathan
 rwatson -> kensmith
 rwatson -> kmacy
 rwatson -> linimon
 rwatson -> rmacklem
 rwatson -> shafeeq
 rwatson -> tmm
 rwatson -> trasz
 rwatson -> trhodes
 rwatson -> wsalamon
 
 sam -> andre
 sam -> benjsc
 sam -> sephe
 
 sbruno -> hiren
 sbruno -> jimharris
 
 schweikh -> dds
 
 scottl -> achim
 scottl -> jimharris
 scottl -> pjd
 scottl -> sah
 scottl -> sbruno
 scottl -> slm
 scottl -> yongari
 
 sheldonh -> dwmalone
 sheldonh -> iedowse
 
 shin -> ume
 
 simon -> benl
 
 sos -> marcel
 
 thompsa -> weongyo
 thompsa -> eri
 
 trasz -> jh
 trasz -> mjg
 
 ume -> jinmei
 ume -> suz
 ume -> tshiozak
 
 wes -> scf
 
 wkoszek -> jceel
 
 wollman -> gad
 
 zml -> mdf
 zml -> zack
 
 }
Index: projects/clang360-import/share
===================================================================
--- projects/clang360-import/share	(revision 277944)
+++ projects/clang360-import/share	(revision 277945)

Property changes on: projects/clang360-import/share
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/share:r277896-277944
Index: projects/clang360-import/sys/arm/broadcom/bcm2835/bcm2835_gpio.c
===================================================================
--- projects/clang360-import/sys/arm/broadcom/bcm2835/bcm2835_gpio.c	(revision 277944)
+++ projects/clang360-import/sys/arm/broadcom/bcm2835/bcm2835_gpio.c	(revision 277945)
@@ -1,792 +1,772 @@
 /*-
  * Copyright (c) 2012 Oleksandr Tymoshenko <gonzo@freebsd.org>
  * Copyright (c) 2012 Luiz Otavio O Souza.
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  *
  */
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/bus.h>
 
 #include <sys/kernel.h>
 #include <sys/module.h>
 #include <sys/rman.h>
 #include <sys/lock.h>
 #include <sys/mutex.h>
 #include <sys/gpio.h>
 #include <sys/sysctl.h>
 
 #include <machine/bus.h>
 #include <machine/cpu.h>
 #include <machine/cpufunc.h>
 #include <machine/resource.h>
 #include <machine/fdt.h>
 #include <machine/intr.h>
 
 #include <dev/fdt/fdt_common.h>
 #include <dev/ofw/ofw_bus.h>
 #include <dev/ofw/ofw_bus_subr.h>
 
 #include <arm/broadcom/bcm2835/bcm2835_gpio.h>
 
 #include "gpio_if.h"
 
 #ifdef DEBUG
 #define dprintf(fmt, args...) do { printf("%s(): ", __func__);   \
     printf(fmt,##args); } while (0)
 #else
 #define dprintf(fmt, args...)
 #endif
 
+#define	BCM_GPIO_IRQS		4
 #define	BCM_GPIO_PINS		54
 #define	BCM_GPIO_DEFAULT_CAPS	(GPIO_PIN_INPUT | GPIO_PIN_OUTPUT |	\
     GPIO_PIN_PULLUP | GPIO_PIN_PULLDOWN)
 
+static struct resource_spec bcm_gpio_res_spec[] = {
+	{ SYS_RES_MEMORY, 0, RF_ACTIVE },
+	{ SYS_RES_IRQ, 0, RF_ACTIVE },
+	{ SYS_RES_IRQ, 1, RF_ACTIVE },
+	{ SYS_RES_IRQ, 2, RF_ACTIVE },
+	{ SYS_RES_IRQ, 3, RF_ACTIVE },
+	{ -1, 0, 0 }
+};
+
 struct bcm_gpio_sysctl {
 	struct bcm_gpio_softc	*sc;
 	uint32_t		pin;
 };
 
 struct bcm_gpio_softc {
 	device_t		sc_dev;
 	struct mtx		sc_mtx;
-	struct resource *	sc_mem_res;
-	struct resource *	sc_irq_res;
+	struct resource *	sc_res[BCM_GPIO_IRQS + 1];
 	bus_space_tag_t		sc_bst;
 	bus_space_handle_t	sc_bsh;
 	void *			sc_intrhand;
 	int			sc_gpio_npins;
 	int			sc_ro_npins;
 	int			sc_ro_pins[BCM_GPIO_PINS];
 	struct gpio_pin		sc_gpio_pins[BCM_GPIO_PINS];
 	struct bcm_gpio_sysctl	sc_sysctl[BCM_GPIO_PINS];
 };
 
 enum bcm_gpio_pud {
 	BCM_GPIO_NONE,
 	BCM_GPIO_PULLDOWN,
 	BCM_GPIO_PULLUP,
 };
 
 #define	BCM_GPIO_LOCK(_sc)	mtx_lock(&_sc->sc_mtx)
 #define	BCM_GPIO_UNLOCK(_sc)	mtx_unlock(&_sc->sc_mtx)
 #define	BCM_GPIO_LOCK_ASSERT(_sc)	mtx_assert(&_sc->sc_mtx, MA_OWNED)
 
 #define	BCM_GPIO_GPFSEL(_bank)	0x00 + _bank * 4
 #define	BCM_GPIO_GPSET(_bank)	0x1c + _bank * 4
 #define	BCM_GPIO_GPCLR(_bank)	0x28 + _bank * 4
 #define	BCM_GPIO_GPLEV(_bank)	0x34 + _bank * 4
 #define	BCM_GPIO_GPPUD(_bank)	0x94
 #define	BCM_GPIO_GPPUDCLK(_bank)	0x98 + _bank * 4
 
 #define	BCM_GPIO_WRITE(_sc, _off, _val)		\
     bus_space_write_4(_sc->sc_bst, _sc->sc_bsh, _off, _val)
 #define	BCM_GPIO_READ(_sc, _off)		\
     bus_space_read_4(_sc->sc_bst, _sc->sc_bsh, _off)
 
 static int
 bcm_gpio_pin_is_ro(struct bcm_gpio_softc *sc, int pin)
 {
 	int i;
 
 	for (i = 0; i < sc->sc_ro_npins; i++)
 		if (pin == sc->sc_ro_pins[i])
 			return (1);
 	return (0);
 }
 
 static uint32_t
 bcm_gpio_get_function(struct bcm_gpio_softc *sc, uint32_t pin)
 {
 	uint32_t bank, func, offset;
 
 	/* Five banks, 10 pins per bank, 3 bits per pin. */
 	bank = pin / 10;
 	offset = (pin - bank * 10) * 3;
 
 	BCM_GPIO_LOCK(sc);
 	func = (BCM_GPIO_READ(sc, BCM_GPIO_GPFSEL(bank)) >> offset) & 7;
 	BCM_GPIO_UNLOCK(sc);
 
 	return (func);
 }
 
 static void
 bcm_gpio_func_str(uint32_t nfunc, char *buf, int bufsize)
 {
 
 	switch (nfunc) {
 	case BCM_GPIO_INPUT:
 		strncpy(buf, "input", bufsize);
 		break;
 	case BCM_GPIO_OUTPUT:
 		strncpy(buf, "output", bufsize);
 		break;
 	case BCM_GPIO_ALT0:
 		strncpy(buf, "alt0", bufsize);
 		break;
 	case BCM_GPIO_ALT1:
 		strncpy(buf, "alt1", bufsize);
 		break;
 	case BCM_GPIO_ALT2:
 		strncpy(buf, "alt2", bufsize);
 		break;
 	case BCM_GPIO_ALT3:
 		strncpy(buf, "alt3", bufsize);
 		break;
 	case BCM_GPIO_ALT4:
 		strncpy(buf, "alt4", bufsize);
 		break;
 	case BCM_GPIO_ALT5:
 		strncpy(buf, "alt5", bufsize);
 		break;
 	default:
 		strncpy(buf, "invalid", bufsize);
 	}
 }
 
 static int
 bcm_gpio_str_func(char *func, uint32_t *nfunc)
 {
 
 	if (strcasecmp(func, "input") == 0)
 		*nfunc = BCM_GPIO_INPUT;
 	else if (strcasecmp(func, "output") == 0)
 		*nfunc = BCM_GPIO_OUTPUT;
 	else if (strcasecmp(func, "alt0") == 0)
 		*nfunc = BCM_GPIO_ALT0;
 	else if (strcasecmp(func, "alt1") == 0)
 		*nfunc = BCM_GPIO_ALT1;
 	else if (strcasecmp(func, "alt2") == 0)
 		*nfunc = BCM_GPIO_ALT2;
 	else if (strcasecmp(func, "alt3") == 0)
 		*nfunc = BCM_GPIO_ALT3;
 	else if (strcasecmp(func, "alt4") == 0)
 		*nfunc = BCM_GPIO_ALT4;
 	else if (strcasecmp(func, "alt5") == 0)
 		*nfunc = BCM_GPIO_ALT5;
 	else
 		return (-1);
 
 	return (0);
 }
 
 static uint32_t
 bcm_gpio_func_flag(uint32_t nfunc)
 {
 
 	switch (nfunc) {
 	case BCM_GPIO_INPUT:
 		return (GPIO_PIN_INPUT);
 	case BCM_GPIO_OUTPUT:
 		return (GPIO_PIN_OUTPUT);
 	}
 	return (0);
 }
 
 static void
 bcm_gpio_set_function(struct bcm_gpio_softc *sc, uint32_t pin, uint32_t f)
 {
 	uint32_t bank, data, offset;
 
 	/* Must be called with lock held. */
 	BCM_GPIO_LOCK_ASSERT(sc);
 
 	/* Five banks, 10 pins per bank, 3 bits per pin. */
 	bank = pin / 10;
 	offset = (pin - bank * 10) * 3;
 
 	data = BCM_GPIO_READ(sc, BCM_GPIO_GPFSEL(bank));
 	data &= ~(7 << offset);
 	data |= (f << offset);
 	BCM_GPIO_WRITE(sc, BCM_GPIO_GPFSEL(bank), data);
 }
 
 static void
 bcm_gpio_set_pud(struct bcm_gpio_softc *sc, uint32_t pin, uint32_t state)
 {
 	uint32_t bank, offset;
 
 	/* Must be called with lock held. */
 	BCM_GPIO_LOCK_ASSERT(sc);
 
 	bank = pin / 32;
 	offset = pin - 32 * bank;
 
 	BCM_GPIO_WRITE(sc, BCM_GPIO_GPPUD(0), state);
 	BCM_GPIO_WRITE(sc, BCM_GPIO_GPPUDCLK(bank), (1 << offset));
 	BCM_GPIO_WRITE(sc, BCM_GPIO_GPPUD(0), 0);
 	BCM_GPIO_WRITE(sc, BCM_GPIO_GPPUDCLK(bank), 0);
 }
 
 void
 bcm_gpio_set_alternate(device_t dev, uint32_t pin, uint32_t nfunc)
 {
 	struct bcm_gpio_softc *sc;
 	int i;
 
 	sc = device_get_softc(dev);
 	BCM_GPIO_LOCK(sc);
 
 	/* Disable pull-up or pull-down on pin. */
 	bcm_gpio_set_pud(sc, pin, BCM_GPIO_NONE);
 
 	/* And now set the pin function. */
 	bcm_gpio_set_function(sc, pin, nfunc);
 
 	/* Update the pin flags. */
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 	if (i < sc->sc_gpio_npins)
 		sc->sc_gpio_pins[i].gp_flags = bcm_gpio_func_flag(nfunc);
 
         BCM_GPIO_UNLOCK(sc);
 }
 
 static void
 bcm_gpio_pin_configure(struct bcm_gpio_softc *sc, struct gpio_pin *pin,
     unsigned int flags)
 {
 
 	BCM_GPIO_LOCK(sc);
 
 	/*
 	 * Manage input/output.
 	 */
 	if (flags & (GPIO_PIN_INPUT|GPIO_PIN_OUTPUT)) {
 		pin->gp_flags &= ~(GPIO_PIN_INPUT|GPIO_PIN_OUTPUT);
 		if (flags & GPIO_PIN_OUTPUT) {
 			pin->gp_flags |= GPIO_PIN_OUTPUT;
 			bcm_gpio_set_function(sc, pin->gp_pin,
 			    BCM_GPIO_OUTPUT);
 		} else {
 			pin->gp_flags |= GPIO_PIN_INPUT;
 			bcm_gpio_set_function(sc, pin->gp_pin,
 			    BCM_GPIO_INPUT);
 		}
 	}
 
 	/* Manage Pull-up/pull-down. */
 	pin->gp_flags &= ~(GPIO_PIN_PULLUP|GPIO_PIN_PULLDOWN);
 	if (flags & (GPIO_PIN_PULLUP|GPIO_PIN_PULLDOWN)) {
 		if (flags & GPIO_PIN_PULLUP) {
 			pin->gp_flags |= GPIO_PIN_PULLUP;
 			bcm_gpio_set_pud(sc, pin->gp_pin, BCM_GPIO_PULLUP);
 		} else {
 			pin->gp_flags |= GPIO_PIN_PULLDOWN;
 			bcm_gpio_set_pud(sc, pin->gp_pin, BCM_GPIO_PULLDOWN);
 		}
 	} else 
 		bcm_gpio_set_pud(sc, pin->gp_pin, BCM_GPIO_NONE);
 
 	BCM_GPIO_UNLOCK(sc);
 }
 
 static int
 bcm_gpio_pin_max(device_t dev, int *maxpin)
 {
 
 	*maxpin = BCM_GPIO_PINS - 1;
 	return (0);
 }
 
 static int
 bcm_gpio_pin_getcaps(device_t dev, uint32_t pin, uint32_t *caps)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	BCM_GPIO_LOCK(sc);
 	*caps = sc->sc_gpio_pins[i].gp_caps;
 	BCM_GPIO_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
 bcm_gpio_pin_getflags(device_t dev, uint32_t pin, uint32_t *flags)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	BCM_GPIO_LOCK(sc);
 	*flags = sc->sc_gpio_pins[i].gp_flags;
 	BCM_GPIO_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
 bcm_gpio_pin_getname(device_t dev, uint32_t pin, char *name)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	BCM_GPIO_LOCK(sc);
 	memcpy(name, sc->sc_gpio_pins[i].gp_name, GPIOMAXNAME);
 	BCM_GPIO_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
 bcm_gpio_pin_setflags(device_t dev, uint32_t pin, uint32_t flags)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	/* We never touch on read-only/reserved pins. */
 	if (bcm_gpio_pin_is_ro(sc, pin))
 		return (EINVAL);
 
 	bcm_gpio_pin_configure(sc, &sc->sc_gpio_pins[i], flags);
 
 	return (0);
 }
 
 static int
 bcm_gpio_pin_set(device_t dev, uint32_t pin, unsigned int value)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	uint32_t bank, offset;
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	/* We never write to read-only/reserved pins. */
 	if (bcm_gpio_pin_is_ro(sc, pin))
 		return (EINVAL);
 
 	bank = pin / 32;
 	offset = pin - 32 * bank;
 
 	BCM_GPIO_LOCK(sc);
 	if (value)
 		BCM_GPIO_WRITE(sc, BCM_GPIO_GPSET(bank), (1 << offset));
 	else
 		BCM_GPIO_WRITE(sc, BCM_GPIO_GPCLR(bank), (1 << offset));
 	BCM_GPIO_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
 bcm_gpio_pin_get(device_t dev, uint32_t pin, unsigned int *val)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	uint32_t bank, offset, reg_data;
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	bank = pin / 32;
 	offset = pin - 32 * bank;
 
 	BCM_GPIO_LOCK(sc);
 	reg_data = BCM_GPIO_READ(sc, BCM_GPIO_GPLEV(bank));
 	BCM_GPIO_UNLOCK(sc);
 	*val = (reg_data & (1 << offset)) ? 1 : 0;
 
 	return (0);
 }
 
 static int
 bcm_gpio_pin_toggle(device_t dev, uint32_t pin)
 {
 	struct bcm_gpio_softc *sc = device_get_softc(dev);
 	uint32_t bank, data, offset;
 	int i;
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 		if (sc->sc_gpio_pins[i].gp_pin == pin)
 			break;
 	}
 
 	if (i >= sc->sc_gpio_npins)
 		return (EINVAL);
 
 	/* We never write to read-only/reserved pins. */
 	if (bcm_gpio_pin_is_ro(sc, pin))
 		return (EINVAL);
 
 	bank = pin / 32;
 	offset = pin - 32 * bank;
 
 	BCM_GPIO_LOCK(sc);
 	data = BCM_GPIO_READ(sc, BCM_GPIO_GPLEV(bank));
 	if (data & (1 << offset))
 		BCM_GPIO_WRITE(sc, BCM_GPIO_GPCLR(bank), (1 << offset));
 	else
 		BCM_GPIO_WRITE(sc, BCM_GPIO_GPSET(bank), (1 << offset));
 	BCM_GPIO_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
-bcm_gpio_get_ro_pins(struct bcm_gpio_softc *sc)
-{
-	int i, len;
-	pcell_t pins[BCM_GPIO_PINS];
-	phandle_t gpio;
-
-	/* Find the gpio node to start. */
-	gpio = ofw_bus_get_node(sc->sc_dev);
-
-	len = OF_getproplen(gpio, "broadcom,read-only");
-	if (len < 0 || len > sizeof(pins))
-		return (-1);
-
-	if (OF_getprop(gpio, "broadcom,read-only", &pins, len) < 0)
-		return (-1);
-
-	sc->sc_ro_npins = len / sizeof(pcell_t);
-
-	device_printf(sc->sc_dev, "read-only pins: ");
-	for (i = 0; i < sc->sc_ro_npins; i++) {
-		sc->sc_ro_pins[i] = fdt32_to_cpu(pins[i]);
-		if (i > 0)
-			printf(",");
-		printf("%d", sc->sc_ro_pins[i]);
-	}
-	if (i > 0)
-		printf(".");
-	printf("\n");
-
-	return (0);
-}
-
-static int
 bcm_gpio_func_proc(SYSCTL_HANDLER_ARGS)
 {
 	char buf[16];
 	struct bcm_gpio_softc *sc;
 	struct bcm_gpio_sysctl *sc_sysctl;
 	uint32_t nfunc;
 	int error;
 
 	sc_sysctl = arg1;
 	sc = sc_sysctl->sc;
 
 	/* Get the current pin function. */
 	nfunc = bcm_gpio_get_function(sc, sc_sysctl->pin);
 	bcm_gpio_func_str(nfunc, buf, sizeof(buf));
 
 	error = sysctl_handle_string(oidp, buf, sizeof(buf), req);
 	if (error != 0 || req->newptr == NULL)
 		return (error);
-
+	/* Ignore changes on read-only pins. */
+	if (bcm_gpio_pin_is_ro(sc, sc_sysctl->pin))
+		return (0);
 	/* Parse the user supplied string and check for a valid pin function. */
 	if (bcm_gpio_str_func(buf, &nfunc) != 0)
 		return (EINVAL);
 
 	/* Update the pin alternate function. */
 	bcm_gpio_set_alternate(sc->sc_dev, sc_sysctl->pin, nfunc);
 
 	return (0);
 }
 
 static void
 bcm_gpio_sysctl_init(struct bcm_gpio_softc *sc)
 {
 	char pinbuf[3];
 	struct bcm_gpio_sysctl *sc_sysctl;
 	struct sysctl_ctx_list *ctx;
 	struct sysctl_oid *tree_node, *pin_node, *pinN_node;
 	struct sysctl_oid_list *tree, *pin_tree, *pinN_tree;
 	int i;
 
 	/*
 	 * Add per-pin sysctl tree/handlers.
 	 */
 	ctx = device_get_sysctl_ctx(sc->sc_dev);
  	tree_node = device_get_sysctl_tree(sc->sc_dev);
  	tree = SYSCTL_CHILDREN(tree_node);
 	pin_node = SYSCTL_ADD_NODE(ctx, tree, OID_AUTO, "pin",
 	    CTLFLAG_RD, NULL, "GPIO Pins");
 	pin_tree = SYSCTL_CHILDREN(pin_node);
 
 	for (i = 0; i < sc->sc_gpio_npins; i++) {
 
 		snprintf(pinbuf, sizeof(pinbuf), "%d", i);
 		pinN_node = SYSCTL_ADD_NODE(ctx, pin_tree, OID_AUTO, pinbuf,
 		    CTLFLAG_RD, NULL, "GPIO Pin");
 		pinN_tree = SYSCTL_CHILDREN(pinN_node);
 
 		sc->sc_sysctl[i].sc = sc;
 		sc_sysctl = &sc->sc_sysctl[i];
 		sc_sysctl->sc = sc;
 		sc_sysctl->pin = sc->sc_gpio_pins[i].gp_pin;
 		SYSCTL_ADD_PROC(ctx, pinN_tree, OID_AUTO, "function",
 		    CTLFLAG_RW | CTLTYPE_STRING, sc_sysctl,
 		    sizeof(struct bcm_gpio_sysctl), bcm_gpio_func_proc,
 		    "A", "Pin Function");
 	}
 }
 
 static int
+bcm_gpio_get_ro_pins(struct bcm_gpio_softc *sc, phandle_t node,
+	const char *propname, const char *label)
+{
+	int i, need_comma, npins, range_start, range_stop;
+	pcell_t *pins;
+
+	/* Get the property data. */
+	npins = OF_getencprop_alloc(node, propname, sizeof(*pins),
+	    (void **)&pins);
+	if (npins < 0)
+		return (-1);
+	if (npins == 0) {
+		free(pins, M_OFWPROP);
+		return (0);
+	}
+	for (i = 0; i < npins; i++)
+		sc->sc_ro_pins[i + sc->sc_ro_npins] = pins[i];
+	sc->sc_ro_npins += npins;
+	need_comma = 0;
+	device_printf(sc->sc_dev, "%s pins: ", label);
+	range_start = range_stop = pins[0];
+	for (i = 1; i < npins; i++) {
+		if (pins[i] != range_stop + 1) {
+			if (need_comma)
+				printf(",");
+			if (range_start != range_stop)
+				printf("%d-%d", range_start, range_stop);
+			else
+				printf("%d", range_start);
+			range_start = range_stop = pins[i];
+			need_comma = 1;
+		} else
+			range_stop++;
+	}
+	if (need_comma)
+		printf(",");
+	if (range_start != range_stop)
+		printf("%d-%d.\n", range_start, range_stop);
+	else
+		printf("%d.\n", range_start);
+	free(pins, M_OFWPROP);
+
+	return (0);
+}
+
+static int
 bcm_gpio_get_reserved_pins(struct bcm_gpio_softc *sc)
 {
-	int i, j, len, npins;
-	pcell_t pins[BCM_GPIO_PINS];
+	char *name;
 	phandle_t gpio, node, reserved;
-	char name[32];
+	ssize_t len;
 
 	/* Get read-only pins. */
-	if (bcm_gpio_get_ro_pins(sc) != 0)
-		return (-1);
-
-	/* Find the gpio/reserved pins node to start. */
 	gpio = ofw_bus_get_node(sc->sc_dev);
-	node = OF_child(gpio);
-	
-	/*
-	 * Find reserved node
-	 */
+	if (bcm_gpio_get_ro_pins(sc, gpio, "broadcom,read-only",
+	    "read-only") != 0)
+		return (-1);
+	/* Traverse the GPIO subnodes to find the reserved pins node. */
 	reserved = 0;
+	node = OF_child(gpio);
 	while ((node != 0) && (reserved == 0)) {
-		len = OF_getprop(node, "name", name,
-		    sizeof(name) - 1);
-		name[len] = 0;
+		len = OF_getprop_alloc(node, "name", 1, (void **)&name);
+		if (len == -1)
+			return (-1);
 		if (strcmp(name, "reserved") == 0)
 			reserved = node;
+		free(name, M_OFWPROP);
 		node = OF_peer(node);
 	}
-
 	if (reserved == 0)
 		return (-1);
-
 	/* Get the reserved pins. */
-	len = OF_getproplen(reserved, "broadcom,pins");
-	if (len < 0 || len > sizeof(pins))
+	if (bcm_gpio_get_ro_pins(sc, reserved, "broadcom,pins",
+	    "reserved") != 0)
 		return (-1);
 
-	if (OF_getprop(reserved, "broadcom,pins", &pins, len) < 0)
-		return (-1);
-
-	npins = len / sizeof(pcell_t);
-
-	j = 0;
-	device_printf(sc->sc_dev, "reserved pins: ");
-	for (i = 0; i < npins; i++) {
-		if (i > 0)
-			printf(",");
-		printf("%d", fdt32_to_cpu(pins[i]));
-		/* Some pins maybe already on the list of read-only pins. */
-		if (bcm_gpio_pin_is_ro(sc, fdt32_to_cpu(pins[i])))
-			continue;
-		sc->sc_ro_pins[j++ + sc->sc_ro_npins] = fdt32_to_cpu(pins[i]);
-	}
-	sc->sc_ro_npins += j;
-	if (i > 0)
-		printf(".");
-	printf("\n");
-
 	return (0);
 }
 
 static int
 bcm_gpio_probe(device_t dev)
 {
 
 	if (!ofw_bus_status_okay(dev))
 		return (ENXIO);
 
 	if (!ofw_bus_is_compatible(dev, "broadcom,bcm2835-gpio"))
 		return (ENXIO);
 
 	device_set_desc(dev, "BCM2708/2835 GPIO controller");
 	return (BUS_PROBE_DEFAULT);
 }
 
 static int
 bcm_gpio_attach(device_t dev)
 {
-	struct bcm_gpio_softc *sc = device_get_softc(dev);
-	uint32_t func;
-	int i, j, rid;
+	int i, j;
 	phandle_t gpio;
+	struct bcm_gpio_softc *sc;
+	uint32_t func;
 
+ 	sc = device_get_softc(dev);
 	sc->sc_dev = dev;
-
 	mtx_init(&sc->sc_mtx, "bcm gpio", "gpio", MTX_DEF);
-
-	rid = 0;
-	sc->sc_mem_res = bus_alloc_resource_any(dev, SYS_RES_MEMORY, &rid,
-	    RF_ACTIVE);
-	if (!sc->sc_mem_res) {
-		device_printf(dev, "cannot allocate memory window\n");
-		return (ENXIO);
+	if (bus_alloc_resources(dev, bcm_gpio_res_spec, sc->sc_res) != 0) {
+		device_printf(dev, "cannot allocate resources\n");
+		goto fail;
 	}
+	sc->sc_bst = rman_get_bustag(sc->sc_res[0]);
+	sc->sc_bsh = rman_get_bushandle(sc->sc_res[0]);
 
-	sc->sc_bst = rman_get_bustag(sc->sc_mem_res);
-	sc->sc_bsh = rman_get_bushandle(sc->sc_mem_res);
-
-	rid = 0;
-	sc->sc_irq_res = bus_alloc_resource_any(dev, SYS_RES_IRQ, &rid,
-	    RF_ACTIVE);
-	if (!sc->sc_irq_res) {
-		bus_release_resource(dev, SYS_RES_MEMORY, 0, sc->sc_mem_res);
-		device_printf(dev, "cannot allocate interrupt\n");
-		return (ENXIO);
-	}
-
 	/* Find our node. */
 	gpio = ofw_bus_get_node(sc->sc_dev);
 
 	if (!OF_hasprop(gpio, "gpio-controller"))
 		/* Node is not a GPIO controller. */
 		goto fail;
 
 	/*
 	 * Find the read-only pins.  These are pins we never touch or bad
 	 * things could happen.
 	 */
 	if (bcm_gpio_get_reserved_pins(sc) == -1)
 		goto fail;
 
 	/* Initialize the software controlled pins. */
 	for (i = 0, j = 0; j < BCM_GPIO_PINS; j++) {
-		if (bcm_gpio_pin_is_ro(sc, j))
-			continue;
 		snprintf(sc->sc_gpio_pins[i].gp_name, GPIOMAXNAME,
 		    "pin %d", j);
 		func = bcm_gpio_get_function(sc, j);
 		sc->sc_gpio_pins[i].gp_pin = j;
 		sc->sc_gpio_pins[i].gp_caps = BCM_GPIO_DEFAULT_CAPS;
 		sc->sc_gpio_pins[i].gp_flags = bcm_gpio_func_flag(func);
 		i++;
 	}
 	sc->sc_gpio_npins = i;
 
 	bcm_gpio_sysctl_init(sc);
 
 	device_add_child(dev, "gpioc", -1);
 	device_add_child(dev, "gpiobus", -1);
 
 	return (bus_generic_attach(dev));
 
 fail:
-	if (sc->sc_irq_res)
-		bus_release_resource(dev, SYS_RES_IRQ, 0, sc->sc_irq_res);
-	if (sc->sc_mem_res)
-		bus_release_resource(dev, SYS_RES_MEMORY, 0, sc->sc_mem_res);
+	bus_release_resources(dev, bcm_gpio_res_spec, sc->sc_res);
+	mtx_destroy(&sc->sc_mtx);
+
 	return (ENXIO);
 }
 
 static int
 bcm_gpio_detach(device_t dev)
 {
 
 	return (EBUSY);
 }
 
 static phandle_t
 bcm_gpio_get_node(device_t bus, device_t dev)
 {
 
 	/* We only have one child, the GPIO bus, which needs our own node. */
 	return (ofw_bus_get_node(bus));
 }
 
 static device_method_t bcm_gpio_methods[] = {
 	/* Device interface */
 	DEVMETHOD(device_probe,		bcm_gpio_probe),
 	DEVMETHOD(device_attach,	bcm_gpio_attach),
 	DEVMETHOD(device_detach,	bcm_gpio_detach),
 
 	/* GPIO protocol */
 	DEVMETHOD(gpio_pin_max,		bcm_gpio_pin_max),
 	DEVMETHOD(gpio_pin_getname,	bcm_gpio_pin_getname),
 	DEVMETHOD(gpio_pin_getflags,	bcm_gpio_pin_getflags),
 	DEVMETHOD(gpio_pin_getcaps,	bcm_gpio_pin_getcaps),
 	DEVMETHOD(gpio_pin_setflags,	bcm_gpio_pin_setflags),
 	DEVMETHOD(gpio_pin_get,		bcm_gpio_pin_get),
 	DEVMETHOD(gpio_pin_set,		bcm_gpio_pin_set),
 	DEVMETHOD(gpio_pin_toggle,	bcm_gpio_pin_toggle),
 
 	/* ofw_bus interface */
 	DEVMETHOD(ofw_bus_get_node,	bcm_gpio_get_node),
 
 	DEVMETHOD_END
 };
 
 static devclass_t bcm_gpio_devclass;
 
 static driver_t bcm_gpio_driver = {
 	"gpio",
 	bcm_gpio_methods,
 	sizeof(struct bcm_gpio_softc),
 };
 
 DRIVER_MODULE(bcm_gpio, simplebus, bcm_gpio_driver, bcm_gpio_devclass, 0, 0);
Index: projects/clang360-import/sys/boot/efi/libefi/efinet.c
===================================================================
--- projects/clang360-import/sys/boot/efi/libefi/efinet.c	(revision 277944)
+++ projects/clang360-import/sys/boot/efi/libefi/efinet.c	(revision 277945)
@@ -1,310 +1,313 @@
 /*-
  * Copyright (c) 2001 Doug Rabson
  * Copyright (c) 2002, 2006 Marcel Moolenaar
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <netinet/in.h>
 #include <netinet/in_systm.h>
 
 #include <stand.h>
 #include <net.h>
 #include <netif.h>
 
 #include <dev_net.c>
 
 #include <efi.h>
 #include <efilib.h>
 
 static EFI_GUID sn_guid = EFI_SIMPLE_NETWORK_PROTOCOL;
 
 static void efinet_end(struct netif *);
 static int efinet_get(struct iodesc *, void *, size_t, time_t);
 static void efinet_init(struct iodesc *, void *);
 static int efinet_match(struct netif *, void *);
 static int efinet_probe(struct netif *, void *);
 static int efinet_put(struct iodesc *, void *, size_t);
 
 struct netif_driver efinetif = {   
 	.netif_bname = "efinet",
 	.netif_match = efinet_match,
 	.netif_probe = efinet_probe,
 	.netif_init = efinet_init,
 	.netif_get = efinet_get,
 	.netif_put = efinet_put,
 	.netif_end = efinet_end,
 	.netif_ifs = NULL,
 	.netif_nifs = 0
 };
 
 #ifdef EFINET_DEBUG
 static void
 dump_mode(EFI_SIMPLE_NETWORK_MODE *mode)
 {
 	int i;
 
 	printf("State                 = %x\n", mode->State);
 	printf("HwAddressSize         = %u\n", mode->HwAddressSize);
 	printf("MediaHeaderSize       = %u\n", mode->MediaHeaderSize);
 	printf("MaxPacketSize         = %u\n", mode->MaxPacketSize);
 	printf("NvRamSize             = %u\n", mode->NvRamSize);
 	printf("NvRamAccessSize       = %u\n", mode->NvRamAccessSize);
 	printf("ReceiveFilterMask     = %x\n", mode->ReceiveFilterMask);
 	printf("ReceiveFilterSetting  = %u\n", mode->ReceiveFilterSetting);
 	printf("MaxMCastFilterCount   = %u\n", mode->MaxMCastFilterCount);
 	printf("MCastFilterCount      = %u\n", mode->MCastFilterCount);
 	printf("MCastFilter           = {");
 	for (i = 0; i < mode->MCastFilterCount; i++)
 		printf(" %s", ether_sprintf(mode->MCastFilter[i].Addr));
 	printf(" }\n");
 	printf("CurrentAddress        = %s\n",
 	    ether_sprintf(mode->CurrentAddress.Addr));
 	printf("BroadcastAddress      = %s\n",
 	    ether_sprintf(mode->BroadcastAddress.Addr));
 	printf("PermanentAddress      = %s\n",
 	    ether_sprintf(mode->PermanentAddress.Addr));
 	printf("IfType                = %u\n", mode->IfType);
 	printf("MacAddressChangeable  = %d\n", mode->MacAddressChangeable);
 	printf("MultipleTxSupported   = %d\n", mode->MultipleTxSupported);
 	printf("MediaPresentSupported = %d\n", mode->MediaPresentSupported);
 	printf("MediaPresent          = %d\n", mode->MediaPresent);
 }
 #endif
 
 static int
 efinet_match(struct netif *nif, void *machdep_hint)
 {
+	struct devdesc *dev = machdep_hint;
 
-	return (1);
+	if (dev->d_unit - 1 == nif->nif_unit)
+		return (1);
+	return(0);
 }
 
 static int
 efinet_probe(struct netif *nif, void *machdep_hint)
 {
 
 	return (0);
 }
 
 static int
 efinet_put(struct iodesc *desc, void *pkt, size_t len)
 {
 	struct netif *nif = desc->io_netif;
 	EFI_SIMPLE_NETWORK *net;
 	EFI_STATUS status;
 	void *buf;
 
 	net = nif->nif_devdata;
 
 	status = net->Transmit(net, 0, len, pkt, 0, 0, 0);
 	if (status != EFI_SUCCESS)
 		return (-1);
 
 	/* Wait for the buffer to be transmitted */
 	do {
 		buf = 0;	/* XXX Is this needed? */
 		status = net->GetStatus(net, 0, &buf);
 		/*
 		 * XXX EFI1.1 and the E1000 card returns a different 
 		 * address than we gave.  Sigh.
 		 */
 	} while (status == EFI_SUCCESS && buf == 0);
 
 	/* XXX How do we deal with status != EFI_SUCCESS now? */
 	return ((status == EFI_SUCCESS) ? len : -1);
 }
 
 static int
 efinet_get(struct iodesc *desc, void *pkt, size_t len, time_t timeout)
 {
 	struct netif *nif = desc->io_netif;
 	EFI_SIMPLE_NETWORK *net;
 	EFI_STATUS status;
 	UINTN bufsz;
 	time_t t;
 	char buf[2048];
 
 	net = nif->nif_devdata;
 
 	t = time(0);
 	while ((time(0) - t) < timeout) {
 		bufsz = sizeof(buf);
 		status = net->Receive(net, 0, &bufsz, buf, 0, 0, 0);
 		if (status == EFI_SUCCESS) {
 			/*
 			 * XXX EFI1.1 and the E1000 card trash our
 			 * workspace if we do not do this silly copy.
 			 * Either they are not respecting the len
 			 * value or do not like the alignment.
 			 */
 			if (bufsz > len)
 				bufsz = len;
 			bcopy(buf, pkt, bufsz);
 			return (bufsz);
 		}
 		if (status != EFI_NOT_READY)
 			return (0);
 	}
 
 	return (0);
 }
 
 static void
 efinet_init(struct iodesc *desc, void *machdep_hint)
 {
 	struct netif *nif = desc->io_netif;
 	EFI_SIMPLE_NETWORK *net;
 	EFI_HANDLE h;
 	EFI_STATUS status;
 
 	h = nif->nif_driver->netif_ifs[nif->nif_unit].dif_private;
 	status = BS->HandleProtocol(h, &sn_guid, (VOID **)&nif->nif_devdata);
 	if (status != EFI_SUCCESS) {
 		printf("net%d: cannot start interface (status=%ld)\n",
 		    nif->nif_unit, (long)status);
 		return;
 	}
 
 	net = nif->nif_devdata;
 	if (net->Mode->State == EfiSimpleNetworkStopped) {
 		status = net->Start(net);
 		if (status != EFI_SUCCESS) {
 			printf("net%d: cannot start interface (status=%ld)\n",
 			    nif->nif_unit, (long)status);
 			return;
 		}
 	}
 
 	if (net->Mode->State != EfiSimpleNetworkInitialized) {
 		status = net->Initialize(net, 0, 0);
 		if (status != EFI_SUCCESS) {
 			printf("net%d: cannot init. interface (status=%ld)\n",
 			    nif->nif_unit, (long)status);
 			return;
 		}
 	}
 
 	if (net->Mode->ReceiveFilterSetting == 0) {
 		UINT32 mask = EFI_SIMPLE_NETWORK_RECEIVE_UNICAST |
 		    EFI_SIMPLE_NETWORK_RECEIVE_BROADCAST;
 
 		status = net->ReceiveFilters(net, mask, 0, FALSE, 0, 0);
 		if (status != EFI_SUCCESS) {
 			printf("net%d: cannot set rx. filters (status=%ld)\n",
 			    nif->nif_unit, (long)status);
 			return;
 		}
 	}
 
 #ifdef EFINET_DEBUG
 	dump_mode(net->Mode);
 #endif
 
 	bcopy(net->Mode->CurrentAddress.Addr, desc->myea, 6);
 	desc->xid = 1;
 }
 
 static void
 efinet_end(struct netif *nif)
 {
 	EFI_SIMPLE_NETWORK *net = nif->nif_devdata; 
 
 	net->Shutdown(net);
 }
 
 static int efinet_dev_init(void);
 static void efinet_dev_print(int);
 
 struct devsw efinet_dev = {
 	.dv_name = "net",
 	.dv_type = DEVT_NET,
 	.dv_init = efinet_dev_init,
 	.dv_strategy = net_strategy,
 	.dv_open = net_open,
 	.dv_close = net_close,
 	.dv_ioctl = noioctl,
 	.dv_print = efinet_dev_print,
 	.dv_cleanup = NULL
 };
 
 static int
 efinet_dev_init()
 {
 	struct netif_dif *dif;
 	struct netif_stats *stats;
 	EFI_HANDLE *handles;
 	EFI_STATUS status;
 	UINTN sz;
 	int err, i, nifs;
 
 	sz = 0;
 	handles = NULL;
 	status = BS->LocateHandle(ByProtocol, &sn_guid, 0, &sz, 0);
 	if (status == EFI_BUFFER_TOO_SMALL) {
 		handles = (EFI_HANDLE *)malloc(sz);
 		status = BS->LocateHandle(ByProtocol, &sn_guid, 0, &sz,
 		    handles);
 		if (EFI_ERROR(status))
 			free(handles);
 	}
 	if (EFI_ERROR(status))
 		return (efi_status_to_errno(status));
 	nifs = sz / sizeof(EFI_HANDLE);
 	err = efi_register_handles(&efinet_dev, handles, NULL, nifs);
 	free(handles);
 	if (err != 0)
 		return (err);
 
 	efinetif.netif_nifs = nifs;
 	efinetif.netif_ifs = calloc(nifs, sizeof(struct netif_dif));
 
 	stats = calloc(nifs, sizeof(struct netif_stats));
 
 	for (i = 0; i < nifs; i++) {
 		dif = &efinetif.netif_ifs[i];
 		dif->dif_unit = i;
 		dif->dif_nsel = 1;
 		dif->dif_stats = &stats[i];
 		dif->dif_private = efi_find_handle(&efinet_dev, i);
 	}
 
 	return (0);
 }
 
 static void
 efinet_dev_print(int verbose)
 {
 	char line[80];
 	EFI_HANDLE h;
 	int unit;
 
 	for (unit = 0, h = efi_find_handle(&efinet_dev, 0);
 	    h != NULL; h = efi_find_handle(&efinet_dev, ++unit)) {
 		sprintf(line, "    %s%d:\n", efinet_dev.dv_name, unit);
 		pager_output(line);
 	}
 }
Index: projects/clang360-import/sys/boot
===================================================================
--- projects/clang360-import/sys/boot	(revision 277944)
+++ projects/clang360-import/sys/boot	(revision 277945)

Property changes on: projects/clang360-import/sys/boot
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/sys/boot:r277777-277944
Index: projects/clang360-import/sys/cam/ctl/ctl.c
===================================================================
--- projects/clang360-import/sys/cam/ctl/ctl.c	(revision 277944)
+++ projects/clang360-import/sys/cam/ctl/ctl.c	(revision 277945)
@@ -1,14211 +1,14281 @@
 /*-
  * Copyright (c) 2003-2009 Silicon Graphics International Corp.
  * Copyright (c) 2012 The FreeBSD Foundation
  * All rights reserved.
  *
  * Portions of this software were developed by Edward Tomasz Napierala
  * under sponsorship from the FreeBSD Foundation.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions, and the following disclaimer,
  *    without modification.
  * 2. Redistributions in binary form must reproduce at minimum a disclaimer
  *    substantially similar to the "NO WARRANTY" disclaimer below
  *    ("Disclaimer") and any redistribution must be conditioned upon
  *    including a substantially similar Disclaimer requirement for further
  *    binary redistribution.
  *
  * NO WARRANTY
  * THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
  * "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
  * LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTIBILITY AND FITNESS FOR
  * A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
  * HOLDERS OR CONTRIBUTORS BE LIABLE FOR SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
  * STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING
  * IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
  * POSSIBILITY OF SUCH DAMAGES.
  *
- * $Id: //depot/users/kenm/FreeBSD-test2/sys/cam/ctl/ctl.c#8 $
+ * $Id$
  */
 /*
  * CAM Target Layer, a SCSI device emulation subsystem.
  *
  * Author: Ken Merry <ken@FreeBSD.org>
  */
 
 #define _CTL_C
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/ctype.h>
 #include <sys/kernel.h>
 #include <sys/types.h>
 #include <sys/kthread.h>
 #include <sys/bio.h>
 #include <sys/fcntl.h>
 #include <sys/lock.h>
 #include <sys/module.h>
 #include <sys/mutex.h>
 #include <sys/condvar.h>
 #include <sys/malloc.h>
 #include <sys/conf.h>
 #include <sys/ioccom.h>
 #include <sys/queue.h>
 #include <sys/sbuf.h>
 #include <sys/smp.h>
 #include <sys/endian.h>
 #include <sys/sysctl.h>
 #include <vm/uma.h>
 
 #include <cam/cam.h>
 #include <cam/scsi/scsi_all.h>
 #include <cam/scsi/scsi_da.h>
 #include <cam/ctl/ctl_io.h>
 #include <cam/ctl/ctl.h>
 #include <cam/ctl/ctl_frontend.h>
 #include <cam/ctl/ctl_frontend_internal.h>
 #include <cam/ctl/ctl_util.h>
 #include <cam/ctl/ctl_backend.h>
 #include <cam/ctl/ctl_ioctl.h>
 #include <cam/ctl/ctl_ha.h>
 #include <cam/ctl/ctl_private.h>
 #include <cam/ctl/ctl_debug.h>
 #include <cam/ctl/ctl_scsi_all.h>
 #include <cam/ctl/ctl_error.h>
 
 struct ctl_softc *control_softc = NULL;
 
 /*
  * Size and alignment macros needed for Copan-specific HA hardware.  These
  * can go away when the HA code is re-written, and uses busdma for any
  * hardware.
  */
 #define	CTL_ALIGN_8B(target, source, type)				\
 	if (((uint32_t)source & 0x7) != 0)				\
 		target = (type)(source + (0x8 - ((uint32_t)source & 0x7)));\
 	else								\
 		target = (type)source;
 
 #define	CTL_SIZE_8B(target, size)					\
 	if ((size & 0x7) != 0)						\
 		target = size + (0x8 - (size & 0x7));			\
 	else								\
 		target = size;
 
 #define CTL_ALIGN_8B_MARGIN	16
 
 /*
  * Template mode pages.
  */
 
 /*
  * Note that these are default values only.  The actual values will be
  * filled in when the user does a mode sense.
  */
 const static struct copan_debugconf_subpage debugconf_page_default = {
 	DBGCNF_PAGE_CODE | SMPH_SPF,	/* page_code */
 	DBGCNF_SUBPAGE_CODE,		/* subpage */
 	{(sizeof(struct copan_debugconf_subpage) - 4) >> 8,
 	 (sizeof(struct copan_debugconf_subpage) - 4) >> 0}, /* page_length */
 	DBGCNF_VERSION,			/* page_version */
 	{CTL_TIME_IO_DEFAULT_SECS>>8,
 	 CTL_TIME_IO_DEFAULT_SECS>>0},	/* ctl_time_io_secs */
 };
 
 const static struct copan_debugconf_subpage debugconf_page_changeable = {
 	DBGCNF_PAGE_CODE | SMPH_SPF,	/* page_code */
 	DBGCNF_SUBPAGE_CODE,		/* subpage */
 	{(sizeof(struct copan_debugconf_subpage) - 4) >> 8,
 	 (sizeof(struct copan_debugconf_subpage) - 4) >> 0}, /* page_length */
 	0,				/* page_version */
 	{0xff,0xff},			/* ctl_time_io_secs */
 };
 
 const static struct scsi_da_rw_recovery_page rw_er_page_default = {
 	/*page_code*/SMS_RW_ERROR_RECOVERY_PAGE,
 	/*page_length*/sizeof(struct scsi_da_rw_recovery_page) - 2,
 	/*byte3*/SMS_RWER_AWRE|SMS_RWER_ARRE,
 	/*read_retry_count*/0,
 	/*correction_span*/0,
 	/*head_offset_count*/0,
 	/*data_strobe_offset_cnt*/0,
 	/*byte8*/SMS_RWER_LBPERE,
 	/*write_retry_count*/0,
 	/*reserved2*/0,
 	/*recovery_time_limit*/{0, 0},
 };
 
 const static struct scsi_da_rw_recovery_page rw_er_page_changeable = {
 	/*page_code*/SMS_RW_ERROR_RECOVERY_PAGE,
 	/*page_length*/sizeof(struct scsi_da_rw_recovery_page) - 2,
 	/*byte3*/0,
 	/*read_retry_count*/0,
 	/*correction_span*/0,
 	/*head_offset_count*/0,
 	/*data_strobe_offset_cnt*/0,
 	/*byte8*/0,
 	/*write_retry_count*/0,
 	/*reserved2*/0,
 	/*recovery_time_limit*/{0, 0},
 };
 
 const static struct scsi_format_page format_page_default = {
 	/*page_code*/SMS_FORMAT_DEVICE_PAGE,
 	/*page_length*/sizeof(struct scsi_format_page) - 2,
 	/*tracks_per_zone*/ {0, 0},
 	/*alt_sectors_per_zone*/ {0, 0},
 	/*alt_tracks_per_zone*/ {0, 0},
 	/*alt_tracks_per_lun*/ {0, 0},
 	/*sectors_per_track*/ {(CTL_DEFAULT_SECTORS_PER_TRACK >> 8) & 0xff,
 			        CTL_DEFAULT_SECTORS_PER_TRACK & 0xff},
 	/*bytes_per_sector*/ {0, 0},
 	/*interleave*/ {0, 0},
 	/*track_skew*/ {0, 0},
 	/*cylinder_skew*/ {0, 0},
 	/*flags*/ SFP_HSEC,
 	/*reserved*/ {0, 0, 0}
 };
 
 const static struct scsi_format_page format_page_changeable = {
 	/*page_code*/SMS_FORMAT_DEVICE_PAGE,
 	/*page_length*/sizeof(struct scsi_format_page) - 2,
 	/*tracks_per_zone*/ {0, 0},
 	/*alt_sectors_per_zone*/ {0, 0},
 	/*alt_tracks_per_zone*/ {0, 0},
 	/*alt_tracks_per_lun*/ {0, 0},
 	/*sectors_per_track*/ {0, 0},
 	/*bytes_per_sector*/ {0, 0},
 	/*interleave*/ {0, 0},
 	/*track_skew*/ {0, 0},
 	/*cylinder_skew*/ {0, 0},
 	/*flags*/ 0,
 	/*reserved*/ {0, 0, 0}
 };
 
 const static struct scsi_rigid_disk_page rigid_disk_page_default = {
 	/*page_code*/SMS_RIGID_DISK_PAGE,
 	/*page_length*/sizeof(struct scsi_rigid_disk_page) - 2,
 	/*cylinders*/ {0, 0, 0},
 	/*heads*/ CTL_DEFAULT_HEADS,
 	/*start_write_precomp*/ {0, 0, 0},
 	/*start_reduced_current*/ {0, 0, 0},
 	/*step_rate*/ {0, 0},
 	/*landing_zone_cylinder*/ {0, 0, 0},
 	/*rpl*/ SRDP_RPL_DISABLED,
 	/*rotational_offset*/ 0,
 	/*reserved1*/ 0,
 	/*rotation_rate*/ {(CTL_DEFAULT_ROTATION_RATE >> 8) & 0xff,
 			   CTL_DEFAULT_ROTATION_RATE & 0xff},
 	/*reserved2*/ {0, 0}
 };
 
 const static struct scsi_rigid_disk_page rigid_disk_page_changeable = {
 	/*page_code*/SMS_RIGID_DISK_PAGE,
 	/*page_length*/sizeof(struct scsi_rigid_disk_page) - 2,
 	/*cylinders*/ {0, 0, 0},
 	/*heads*/ 0,
 	/*start_write_precomp*/ {0, 0, 0},
 	/*start_reduced_current*/ {0, 0, 0},
 	/*step_rate*/ {0, 0},
 	/*landing_zone_cylinder*/ {0, 0, 0},
 	/*rpl*/ 0,
 	/*rotational_offset*/ 0,
 	/*reserved1*/ 0,
 	/*rotation_rate*/ {0, 0},
 	/*reserved2*/ {0, 0}
 };
 
 const static struct scsi_caching_page caching_page_default = {
 	/*page_code*/SMS_CACHING_PAGE,
 	/*page_length*/sizeof(struct scsi_caching_page) - 2,
 	/*flags1*/ SCP_DISC | SCP_WCE,
 	/*ret_priority*/ 0,
 	/*disable_pf_transfer_len*/ {0xff, 0xff},
 	/*min_prefetch*/ {0, 0},
 	/*max_prefetch*/ {0xff, 0xff},
 	/*max_pf_ceiling*/ {0xff, 0xff},
 	/*flags2*/ 0,
 	/*cache_segments*/ 0,
 	/*cache_seg_size*/ {0, 0},
 	/*reserved*/ 0,
 	/*non_cache_seg_size*/ {0, 0, 0}
 };
 
 const static struct scsi_caching_page caching_page_changeable = {
 	/*page_code*/SMS_CACHING_PAGE,
 	/*page_length*/sizeof(struct scsi_caching_page) - 2,
 	/*flags1*/ SCP_WCE | SCP_RCD,
 	/*ret_priority*/ 0,
 	/*disable_pf_transfer_len*/ {0, 0},
 	/*min_prefetch*/ {0, 0},
 	/*max_prefetch*/ {0, 0},
 	/*max_pf_ceiling*/ {0, 0},
 	/*flags2*/ 0,
 	/*cache_segments*/ 0,
 	/*cache_seg_size*/ {0, 0},
 	/*reserved*/ 0,
 	/*non_cache_seg_size*/ {0, 0, 0}
 };
 
 const static struct scsi_control_page control_page_default = {
 	/*page_code*/SMS_CONTROL_MODE_PAGE,
 	/*page_length*/sizeof(struct scsi_control_page) - 2,
 	/*rlec*/0,
 	/*queue_flags*/SCP_QUEUE_ALG_RESTRICTED,
 	/*eca_and_aen*/0,
 	/*flags4*/SCP_TAS,
 	/*aen_holdoff_period*/{0, 0},
 	/*busy_timeout_period*/{0, 0},
 	/*extended_selftest_completion_time*/{0, 0}
 };
 
 const static struct scsi_control_page control_page_changeable = {
 	/*page_code*/SMS_CONTROL_MODE_PAGE,
 	/*page_length*/sizeof(struct scsi_control_page) - 2,
 	/*rlec*/SCP_DSENSE,
 	/*queue_flags*/SCP_QUEUE_ALG_MASK,
 	/*eca_and_aen*/SCP_SWP,
 	/*flags4*/0,
 	/*aen_holdoff_period*/{0, 0},
 	/*busy_timeout_period*/{0, 0},
 	/*extended_selftest_completion_time*/{0, 0}
 };
 
 const static struct scsi_info_exceptions_page ie_page_default = {
 	/*page_code*/SMS_INFO_EXCEPTIONS_PAGE,
 	/*page_length*/sizeof(struct scsi_info_exceptions_page) - 2,
 	/*info_flags*/SIEP_FLAGS_DEXCPT,
 	/*mrie*/0,
 	/*interval_timer*/{0, 0, 0, 0},
 	/*report_count*/{0, 0, 0, 0}
 };
 
 const static struct scsi_info_exceptions_page ie_page_changeable = {
 	/*page_code*/SMS_INFO_EXCEPTIONS_PAGE,
 	/*page_length*/sizeof(struct scsi_info_exceptions_page) - 2,
 	/*info_flags*/0,
 	/*mrie*/0,
 	/*interval_timer*/{0, 0, 0, 0},
 	/*report_count*/{0, 0, 0, 0}
 };
 
 #define CTL_LBPM_LEN	(sizeof(struct ctl_logical_block_provisioning_page) - 4)
 
 const static struct ctl_logical_block_provisioning_page lbp_page_default = {{
 	/*page_code*/SMS_INFO_EXCEPTIONS_PAGE | SMPH_SPF,
 	/*subpage_code*/0x02,
 	/*page_length*/{CTL_LBPM_LEN >> 8, CTL_LBPM_LEN},
 	/*flags*/0,
 	/*reserved*/{0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0},
 	/*descr*/{}},
 	{{/*flags*/0,
 	  /*resource*/0x01,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}},
 	 {/*flags*/0,
 	  /*resource*/0x02,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}},
 	 {/*flags*/0,
 	  /*resource*/0xf1,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}},
 	 {/*flags*/0,
 	  /*resource*/0xf2,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}}
 	}
 };
 
 const static struct ctl_logical_block_provisioning_page lbp_page_changeable = {{
 	/*page_code*/SMS_INFO_EXCEPTIONS_PAGE | SMPH_SPF,
 	/*subpage_code*/0x02,
 	/*page_length*/{CTL_LBPM_LEN >> 8, CTL_LBPM_LEN},
 	/*flags*/0,
 	/*reserved*/{0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0},
 	/*descr*/{}},
 	{{/*flags*/0,
 	  /*resource*/0,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}},
 	 {/*flags*/0,
 	  /*resource*/0,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}},
 	 {/*flags*/0,
 	  /*resource*/0,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}},
 	 {/*flags*/0,
 	  /*resource*/0,
 	  /*reserved*/{0, 0},
 	  /*count*/{0, 0, 0, 0}}
 	}
 };
 
 /*
  * XXX KDM move these into the softc.
  */
 static int rcv_sync_msg;
 static uint8_t ctl_pause_rtr;
 
 SYSCTL_NODE(_kern_cam, OID_AUTO, ctl, CTLFLAG_RD, 0, "CAM Target Layer");
 static int worker_threads = -1;
 SYSCTL_INT(_kern_cam_ctl, OID_AUTO, worker_threads, CTLFLAG_RDTUN,
     &worker_threads, 1, "Number of worker threads");
 static int ctl_debug = CTL_DEBUG_NONE;
 SYSCTL_INT(_kern_cam_ctl, OID_AUTO, debug, CTLFLAG_RWTUN,
     &ctl_debug, 0, "Enabled debug flags");
 
 /*
  * Supported pages (0x00), Serial number (0x80), Device ID (0x83),
  * Extended INQUIRY Data (0x86), Mode Page Policy (0x87),
  * SCSI Ports (0x88), Third-party Copy (0x8F), Block limits (0xB0),
  * Block Device Characteristics (0xB1) and Logical Block Provisioning (0xB2)
  */
 #define SCSI_EVPD_NUM_SUPPORTED_PAGES	10
 
 static void ctl_isc_event_handler(ctl_ha_channel chanel, ctl_ha_event event,
 				  int param);
 static void ctl_copy_sense_data(union ctl_ha_msg *src, union ctl_io *dest);
 static int ctl_init(void);
 void ctl_shutdown(void);
 static int ctl_open(struct cdev *dev, int flags, int fmt, struct thread *td);
 static int ctl_close(struct cdev *dev, int flags, int fmt, struct thread *td);
 static void ctl_ioctl_online(void *arg);
 static void ctl_ioctl_offline(void *arg);
 static int ctl_ioctl_lun_enable(void *arg, struct ctl_id targ_id, int lun_id);
 static int ctl_ioctl_lun_disable(void *arg, struct ctl_id targ_id, int lun_id);
 static int ctl_ioctl_do_datamove(struct ctl_scsiio *ctsio);
 static int ctl_serialize_other_sc_cmd(struct ctl_scsiio *ctsio);
 static int ctl_ioctl_submit_wait(union ctl_io *io);
 static void ctl_ioctl_datamove(union ctl_io *io);
 static void ctl_ioctl_done(union ctl_io *io);
 static void ctl_ioctl_hard_startstop_callback(void *arg,
 					      struct cfi_metatask *metatask);
 static void ctl_ioctl_bbrread_callback(void *arg,struct cfi_metatask *metatask);
 static int ctl_ioctl_fill_ooa(struct ctl_lun *lun, uint32_t *cur_fill_num,
 			      struct ctl_ooa *ooa_hdr,
 			      struct ctl_ooa_entry *kern_entries);
 static int ctl_ioctl(struct cdev *dev, u_long cmd, caddr_t addr, int flag,
 		     struct thread *td);
 static uint32_t ctl_map_lun(struct ctl_softc *softc, int port_num, uint32_t lun);
 static uint32_t ctl_map_lun_back(struct ctl_softc *softc, int port_num, uint32_t lun);
 static int ctl_alloc_lun(struct ctl_softc *ctl_softc, struct ctl_lun *lun,
 			 struct ctl_be_lun *be_lun, struct ctl_id target_id);
 static int ctl_free_lun(struct ctl_lun *lun);
 static void ctl_create_lun(struct ctl_be_lun *be_lun);
 /**
 static void ctl_failover_change_pages(struct ctl_softc *softc,
 				      struct ctl_scsiio *ctsio, int master);
 **/
 
 static int ctl_do_mode_select(union ctl_io *io);
 static int ctl_pro_preempt(struct ctl_softc *softc, struct ctl_lun *lun,
 			   uint64_t res_key, uint64_t sa_res_key,
 			   uint8_t type, uint32_t residx,
 			   struct ctl_scsiio *ctsio,
 			   struct scsi_per_res_out *cdb,
 			   struct scsi_per_res_out_parms* param);
 static void ctl_pro_preempt_other(struct ctl_lun *lun,
 				  union ctl_ha_msg *msg);
 static void ctl_hndl_per_res_out_on_other_sc(union ctl_ha_msg *msg);
 static int ctl_inquiry_evpd_supported(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd_serial(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd_devid(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd_eid(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd_mpp(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd_scsi_ports(struct ctl_scsiio *ctsio,
 					 int alloc_len);
 static int ctl_inquiry_evpd_block_limits(struct ctl_scsiio *ctsio,
 					 int alloc_len);
 static int ctl_inquiry_evpd_bdc(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd_lbp(struct ctl_scsiio *ctsio, int alloc_len);
 static int ctl_inquiry_evpd(struct ctl_scsiio *ctsio);
 static int ctl_inquiry_std(struct ctl_scsiio *ctsio);
 static int ctl_get_lba_len(union ctl_io *io, uint64_t *lba, uint64_t *len);
 static ctl_action ctl_extent_check(union ctl_io *io1, union ctl_io *io2,
     bool seq);
 static ctl_action ctl_extent_check_seq(union ctl_io *io1, union ctl_io *io2);
 static ctl_action ctl_check_for_blockage(struct ctl_lun *lun,
     union ctl_io *pending_io, union ctl_io *ooa_io);
 static ctl_action ctl_check_ooa(struct ctl_lun *lun, union ctl_io *pending_io,
 				union ctl_io *starting_io);
 static int ctl_check_blocked(struct ctl_lun *lun);
 static int ctl_scsiio_lun_check(struct ctl_lun *lun,
 				const struct ctl_cmd_entry *entry,
 				struct ctl_scsiio *ctsio);
 //static int ctl_check_rtr(union ctl_io *pending_io, struct ctl_softc *softc);
 static void ctl_failover(void);
+static void ctl_clear_ua(struct ctl_softc *ctl_softc, uint32_t initidx,
+			 ctl_ua_type ua_type);
 static int ctl_scsiio_precheck(struct ctl_softc *ctl_softc,
 			       struct ctl_scsiio *ctsio);
 static int ctl_scsiio(struct ctl_scsiio *ctsio);
 
 static int ctl_bus_reset(struct ctl_softc *ctl_softc, union ctl_io *io);
 static int ctl_target_reset(struct ctl_softc *ctl_softc, union ctl_io *io,
 			    ctl_ua_type ua_type);
 static int ctl_lun_reset(struct ctl_lun *lun, union ctl_io *io,
 			 ctl_ua_type ua_type);
 static int ctl_abort_task(union ctl_io *io);
 static int ctl_abort_task_set(union ctl_io *io);
 static int ctl_i_t_nexus_reset(union ctl_io *io);
 static void ctl_run_task(union ctl_io *io);
 #ifdef CTL_IO_DELAY
 static void ctl_datamove_timer_wakeup(void *arg);
 static void ctl_done_timer_wakeup(void *arg);
 #endif /* CTL_IO_DELAY */
 
 static void ctl_send_datamove_done(union ctl_io *io, int have_lock);
 static void ctl_datamove_remote_write_cb(struct ctl_ha_dt_req *rq);
 static int ctl_datamove_remote_dm_write_cb(union ctl_io *io);
 static void ctl_datamove_remote_write(union ctl_io *io);
 static int ctl_datamove_remote_dm_read_cb(union ctl_io *io);
 static void ctl_datamove_remote_read_cb(struct ctl_ha_dt_req *rq);
 static int ctl_datamove_remote_sgl_setup(union ctl_io *io);
 static int ctl_datamove_remote_xfer(union ctl_io *io, unsigned command,
 				    ctl_ha_dt_cb callback);
 static void ctl_datamove_remote_read(union ctl_io *io);
 static void ctl_datamove_remote(union ctl_io *io);
 static int ctl_process_done(union ctl_io *io);
 static void ctl_lun_thread(void *arg);
 static void ctl_thresh_thread(void *arg);
 static void ctl_work_thread(void *arg);
 static void ctl_enqueue_incoming(union ctl_io *io);
 static void ctl_enqueue_rtr(union ctl_io *io);
 static void ctl_enqueue_done(union ctl_io *io);
 static void ctl_enqueue_isc(union ctl_io *io);
 static const struct ctl_cmd_entry *
     ctl_get_cmd_entry(struct ctl_scsiio *ctsio, int *sa);
 static const struct ctl_cmd_entry *
     ctl_validate_command(struct ctl_scsiio *ctsio);
 static int ctl_cmd_applicable(uint8_t lun_type,
     const struct ctl_cmd_entry *entry);
 
 /*
  * Load the serialization table.  This isn't very pretty, but is probably
  * the easiest way to do it.
  */
 #include "ctl_ser_table.c"
 
 /*
  * We only need to define open, close and ioctl routines for this driver.
  */
 static struct cdevsw ctl_cdevsw = {
 	.d_version =	D_VERSION,
 	.d_flags =	0,
 	.d_open =	ctl_open,
 	.d_close =	ctl_close,
 	.d_ioctl =	ctl_ioctl,
 	.d_name =	"ctl",
 };
 
 
 MALLOC_DEFINE(M_CTL, "ctlmem", "Memory used for CTL");
 MALLOC_DEFINE(M_CTLIO, "ctlio", "Memory used for CTL requests");
 
 static int ctl_module_event_handler(module_t, int /*modeventtype_t*/, void *);
 
 static moduledata_t ctl_moduledata = {
 	"ctl",
 	ctl_module_event_handler,
 	NULL
 };
 
 DECLARE_MODULE(ctl, ctl_moduledata, SI_SUB_CONFIGURE, SI_ORDER_THIRD);
 MODULE_VERSION(ctl, 1);
 
 static struct ctl_frontend ioctl_frontend =
 {
 	.name = "ioctl",
 };
 
 static void
 ctl_isc_handler_finish_xfer(struct ctl_softc *ctl_softc,
 			    union ctl_ha_msg *msg_info)
 {
 	struct ctl_scsiio *ctsio;
 
 	if (msg_info->hdr.original_sc == NULL) {
 		printf("%s: original_sc == NULL!\n", __func__);
 		/* XXX KDM now what? */
 		return;
 	}
 
 	ctsio = &msg_info->hdr.original_sc->scsiio;
 	ctsio->io_hdr.flags |= CTL_FLAG_IO_ACTIVE;
 	ctsio->io_hdr.msg_type = CTL_MSG_FINISH_IO;
 	ctsio->io_hdr.status = msg_info->hdr.status;
 	ctsio->scsi_status = msg_info->scsi.scsi_status;
 	ctsio->sense_len = msg_info->scsi.sense_len;
 	ctsio->sense_residual = msg_info->scsi.sense_residual;
 	ctsio->residual = msg_info->scsi.residual;
 	memcpy(&ctsio->sense_data, &msg_info->scsi.sense_data,
 	       sizeof(ctsio->sense_data));
 	memcpy(&ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN].bytes,
 	       &msg_info->scsi.lbalen, sizeof(msg_info->scsi.lbalen));
 	ctl_enqueue_isc((union ctl_io *)ctsio);
 }
 
 static void
 ctl_isc_handler_finish_ser_only(struct ctl_softc *ctl_softc,
 				union ctl_ha_msg *msg_info)
 {
 	struct ctl_scsiio *ctsio;
 
 	if (msg_info->hdr.serializing_sc == NULL) {
 		printf("%s: serializing_sc == NULL!\n", __func__);
 		/* XXX KDM now what? */
 		return;
 	}
 
 	ctsio = &msg_info->hdr.serializing_sc->scsiio;
 #if 0
 	/*
 	 * Attempt to catch the situation where an I/O has
 	 * been freed, and we're using it again.
 	 */
 	if (ctsio->io_hdr.io_type == 0xff) {
 		union ctl_io *tmp_io;
 		tmp_io = (union ctl_io *)ctsio;
 		printf("%s: %p use after free!\n", __func__,
 		       ctsio);
 		printf("%s: type %d msg %d cdb %x iptl: "
 		       "%d:%d:%d:%d tag 0x%04x "
 		       "flag %#x status %x\n",
 			__func__,
 			tmp_io->io_hdr.io_type,
 			tmp_io->io_hdr.msg_type,
 			tmp_io->scsiio.cdb[0],
 			tmp_io->io_hdr.nexus.initid.id,
 			tmp_io->io_hdr.nexus.targ_port,
 			tmp_io->io_hdr.nexus.targ_target.id,
 			tmp_io->io_hdr.nexus.targ_lun,
 			(tmp_io->io_hdr.io_type ==
 			CTL_IO_TASK) ?
 			tmp_io->taskio.tag_num :
 			tmp_io->scsiio.tag_num,
 		        tmp_io->io_hdr.flags,
 			tmp_io->io_hdr.status);
 	}
 #endif
 	ctsio->io_hdr.msg_type = CTL_MSG_FINISH_IO;
 	ctl_enqueue_isc((union ctl_io *)ctsio);
 }
 
 /*
  * ISC (Inter Shelf Communication) event handler.  Events from the HA
  * subsystem come in here.
  */
 static void
 ctl_isc_event_handler(ctl_ha_channel channel, ctl_ha_event event, int param)
 {
 	struct ctl_softc *softc;
 	union ctl_io *io;
 	struct ctl_prio *presio;
 	ctl_ha_status isc_status;
 
 	softc = control_softc;
 	io = NULL;
 
 
 #if 0
 	printf("CTL: Isc Msg event %d\n", event);
 #endif
 	if (event == CTL_HA_EVT_MSG_RECV) {
 		union ctl_ha_msg msg_info;
 
 		isc_status = ctl_ha_msg_recv(CTL_HA_CHAN_CTL, &msg_info,
 					     sizeof(msg_info), /*wait*/ 0);
 #if 0
 		printf("CTL: msg_type %d\n", msg_info.msg_type);
 #endif
 		if (isc_status != 0) {
 			printf("Error receiving message, status = %d\n",
 			       isc_status);
 			return;
 		}
 
 		switch (msg_info.hdr.msg_type) {
 		case CTL_MSG_SERIALIZE:
 #if 0
 			printf("Serialize\n");
 #endif
 			io = ctl_alloc_io_nowait(softc->othersc_pool);
 			if (io == NULL) {
 				printf("ctl_isc_event_handler: can't allocate "
 				       "ctl_io!\n");
 				/* Bad Juju */
 				/* Need to set busy and send msg back */
 				msg_info.hdr.msg_type = CTL_MSG_BAD_JUJU;
 				msg_info.hdr.status = CTL_SCSI_ERROR;
 				msg_info.scsi.scsi_status = SCSI_STATUS_BUSY;
 				msg_info.scsi.sense_len = 0;
 			        if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 				    sizeof(msg_info), 0) > CTL_HA_STATUS_SUCCESS){
 				}
 				goto bailout;
 			}
 			ctl_zero_io(io);
 			// populate ctsio from msg_info
 			io->io_hdr.io_type = CTL_IO_SCSI;
 			io->io_hdr.msg_type = CTL_MSG_SERIALIZE;
 			io->io_hdr.original_sc = msg_info.hdr.original_sc;
 #if 0
 			printf("pOrig %x\n", (int)msg_info.original_sc);
 #endif
 			io->io_hdr.flags |= CTL_FLAG_FROM_OTHER_SC |
 					    CTL_FLAG_IO_ACTIVE;
 			/*
 			 * If we're in serialization-only mode, we don't
 			 * want to go through full done processing.  Thus
 			 * the COPY flag.
 			 *
 			 * XXX KDM add another flag that is more specific.
 			 */
 			if (softc->ha_mode == CTL_HA_MODE_SER_ONLY)
 				io->io_hdr.flags |= CTL_FLAG_INT_COPY;
 			io->io_hdr.nexus = msg_info.hdr.nexus;
 #if 0
 			printf("targ %d, port %d, iid %d, lun %d\n",
 			       io->io_hdr.nexus.targ_target.id,
 			       io->io_hdr.nexus.targ_port,
 			       io->io_hdr.nexus.initid.id,
 			       io->io_hdr.nexus.targ_lun);
 #endif
 			io->scsiio.tag_num = msg_info.scsi.tag_num;
 			io->scsiio.tag_type = msg_info.scsi.tag_type;
 			memcpy(io->scsiio.cdb, msg_info.scsi.cdb,
 			       CTL_MAX_CDBLEN);
 			if (softc->ha_mode == CTL_HA_MODE_XFER) {
 				const struct ctl_cmd_entry *entry;
 
 				entry = ctl_get_cmd_entry(&io->scsiio, NULL);
 				io->io_hdr.flags &= ~CTL_FLAG_DATA_MASK;
 				io->io_hdr.flags |=
 					entry->flags & CTL_FLAG_DATA_MASK;
 			}
 			ctl_enqueue_isc(io);
 			break;
 
 		/* Performed on the Originating SC, XFER mode only */
 		case CTL_MSG_DATAMOVE: {
 			struct ctl_sg_entry *sgl;
 			int i, j;
 
 			io = msg_info.hdr.original_sc;
 			if (io == NULL) {
 				printf("%s: original_sc == NULL!\n", __func__);
 				/* XXX KDM do something here */
 				break;
 			}
 			io->io_hdr.msg_type = CTL_MSG_DATAMOVE;
 			io->io_hdr.flags |= CTL_FLAG_IO_ACTIVE;
 			/*
 			 * Keep track of this, we need to send it back over
 			 * when the datamove is complete.
 			 */
 			io->io_hdr.serializing_sc = msg_info.hdr.serializing_sc;
 
 			if (msg_info.dt.sg_sequence == 0) {
 				/*
 				 * XXX KDM we use the preallocated S/G list
 				 * here, but we'll need to change this to
 				 * dynamic allocation if we need larger S/G
 				 * lists.
 				 */
 				if (msg_info.dt.kern_sg_entries >
 				    sizeof(io->io_hdr.remote_sglist) /
 				    sizeof(io->io_hdr.remote_sglist[0])) {
 					printf("%s: number of S/G entries "
 					    "needed %u > allocated num %zd\n",
 					    __func__,
 					    msg_info.dt.kern_sg_entries,
 					    sizeof(io->io_hdr.remote_sglist)/
 					    sizeof(io->io_hdr.remote_sglist[0]));
 				
 					/*
 					 * XXX KDM send a message back to
 					 * the other side to shut down the
 					 * DMA.  The error will come back
 					 * through via the normal channel.
 					 */
 					break;
 				}
 				sgl = io->io_hdr.remote_sglist;
 				memset(sgl, 0,
 				       sizeof(io->io_hdr.remote_sglist));
 
 				io->scsiio.kern_data_ptr = (uint8_t *)sgl;
 
 				io->scsiio.kern_sg_entries =
 					msg_info.dt.kern_sg_entries;
 				io->scsiio.rem_sg_entries =
 					msg_info.dt.kern_sg_entries;
 				io->scsiio.kern_data_len =
 					msg_info.dt.kern_data_len;
 				io->scsiio.kern_total_len =
 					msg_info.dt.kern_total_len;
 				io->scsiio.kern_data_resid =
 					msg_info.dt.kern_data_resid;
 				io->scsiio.kern_rel_offset =
 					msg_info.dt.kern_rel_offset;
 				/*
 				 * Clear out per-DMA flags.
 				 */
 				io->io_hdr.flags &= ~CTL_FLAG_RDMA_MASK;
 				/*
 				 * Add per-DMA flags that are set for this
 				 * particular DMA request.
 				 */
 				io->io_hdr.flags |= msg_info.dt.flags &
 						    CTL_FLAG_RDMA_MASK;
 			} else
 				sgl = (struct ctl_sg_entry *)
 					io->scsiio.kern_data_ptr;
 
 			for (i = msg_info.dt.sent_sg_entries, j = 0;
 			     i < (msg_info.dt.sent_sg_entries +
 			     msg_info.dt.cur_sg_entries); i++, j++) {
 				sgl[i].addr = msg_info.dt.sg_list[j].addr;
 				sgl[i].len = msg_info.dt.sg_list[j].len;
 
 #if 0
 				printf("%s: L: %p,%d -> %p,%d j=%d, i=%d\n",
 				       __func__,
 				       msg_info.dt.sg_list[j].addr,
 				       msg_info.dt.sg_list[j].len,
 				       sgl[i].addr, sgl[i].len, j, i);
 #endif
 			}
 #if 0
 			memcpy(&sgl[msg_info.dt.sent_sg_entries],
 			       msg_info.dt.sg_list,
 			       sizeof(*sgl) * msg_info.dt.cur_sg_entries);
 #endif
 
 			/*
 			 * If this is the last piece of the I/O, we've got
 			 * the full S/G list.  Queue processing in the thread.
 			 * Otherwise wait for the next piece.
 			 */
 			if (msg_info.dt.sg_last != 0)
 				ctl_enqueue_isc(io);
 			break;
 		}
 		/* Performed on the Serializing (primary) SC, XFER mode only */
 		case CTL_MSG_DATAMOVE_DONE: {
 			if (msg_info.hdr.serializing_sc == NULL) {
 				printf("%s: serializing_sc == NULL!\n",
 				       __func__);
 				/* XXX KDM now what? */
 				break;
 			}
 			/*
 			 * We grab the sense information here in case
 			 * there was a failure, so we can return status
 			 * back to the initiator.
 			 */
 			io = msg_info.hdr.serializing_sc;
 			io->io_hdr.msg_type = CTL_MSG_DATAMOVE_DONE;
 			io->io_hdr.status = msg_info.hdr.status;
 			io->scsiio.scsi_status = msg_info.scsi.scsi_status;
 			io->scsiio.sense_len = msg_info.scsi.sense_len;
 			io->scsiio.sense_residual =msg_info.scsi.sense_residual;
 			io->io_hdr.port_status = msg_info.scsi.fetd_status;
 			io->scsiio.residual = msg_info.scsi.residual;
 			memcpy(&io->scsiio.sense_data,&msg_info.scsi.sense_data,
 			       sizeof(io->scsiio.sense_data));
 			ctl_enqueue_isc(io);
 			break;
 		}
 
 		/* Preformed on Originating SC, SER_ONLY mode */
 		case CTL_MSG_R2R:
 			io = msg_info.hdr.original_sc;
 			if (io == NULL) {
 				printf("%s: Major Bummer\n", __func__);
 				return;
 			} else {
 #if 0
 				printf("pOrig %x\n",(int) ctsio);
 #endif
 			}
 			io->io_hdr.msg_type = CTL_MSG_R2R;
 			io->io_hdr.serializing_sc = msg_info.hdr.serializing_sc;
 			ctl_enqueue_isc(io);
 			break;
 
 		/*
 		 * Performed on Serializing(i.e. primary SC) SC in SER_ONLY
 		 * mode.
 		 * Performed on the Originating (i.e. secondary) SC in XFER
 		 * mode
 		 */
 		case CTL_MSG_FINISH_IO:
 			if (softc->ha_mode == CTL_HA_MODE_XFER)
 				ctl_isc_handler_finish_xfer(softc,
 							    &msg_info);
 			else
 				ctl_isc_handler_finish_ser_only(softc,
 								&msg_info);
 			break;
 
 		/* Preformed on Originating SC */
 		case CTL_MSG_BAD_JUJU:
 			io = msg_info.hdr.original_sc;
 			if (io == NULL) {
 				printf("%s: Bad JUJU!, original_sc is NULL!\n",
 				       __func__);
 				break;
 			}
 			ctl_copy_sense_data(&msg_info, io);
 			/*
 			 * IO should have already been cleaned up on other
 			 * SC so clear this flag so we won't send a message
 			 * back to finish the IO there.
 			 */
 			io->io_hdr.flags &= ~CTL_FLAG_SENT_2OTHER_SC;
 			io->io_hdr.flags |= CTL_FLAG_IO_ACTIVE;
 
 			/* io = msg_info.hdr.serializing_sc; */
 			io->io_hdr.msg_type = CTL_MSG_BAD_JUJU;
 			ctl_enqueue_isc(io);
 			break;
 
 		/* Handle resets sent from the other side */
 		case CTL_MSG_MANAGE_TASKS: {
 			struct ctl_taskio *taskio;
 			taskio = (struct ctl_taskio *)ctl_alloc_io_nowait(
 			    softc->othersc_pool);
 			if (taskio == NULL) {
 				printf("ctl_isc_event_handler: can't allocate "
 				       "ctl_io!\n");
 				/* Bad Juju */
 				/* should I just call the proper reset func
 				   here??? */
 				goto bailout;
 			}
 			ctl_zero_io((union ctl_io *)taskio);
 			taskio->io_hdr.io_type = CTL_IO_TASK;
 			taskio->io_hdr.flags |= CTL_FLAG_FROM_OTHER_SC;
 			taskio->io_hdr.nexus = msg_info.hdr.nexus;
 			taskio->task_action = msg_info.task.task_action;
 			taskio->tag_num = msg_info.task.tag_num;
 			taskio->tag_type = msg_info.task.tag_type;
 #ifdef CTL_TIME_IO
 			taskio->io_hdr.start_time = time_uptime;
 			getbintime(&taskio->io_hdr.start_bt);
 #if 0
 			cs_prof_gettime(&taskio->io_hdr.start_ticks);
 #endif
 #endif /* CTL_TIME_IO */
 			ctl_run_task((union ctl_io *)taskio);
 			break;
 		}
 		/* Persistent Reserve action which needs attention */
 		case CTL_MSG_PERS_ACTION:
 			presio = (struct ctl_prio *)ctl_alloc_io_nowait(
 			    softc->othersc_pool);
 			if (presio == NULL) {
 				printf("ctl_isc_event_handler: can't allocate "
 				       "ctl_io!\n");
 				/* Bad Juju */
 				/* Need to set busy and send msg back */
 				goto bailout;
 			}
 			ctl_zero_io((union ctl_io *)presio);
 			presio->io_hdr.msg_type = CTL_MSG_PERS_ACTION;
 			presio->pr_msg = msg_info.pr;
 			ctl_enqueue_isc((union ctl_io *)presio);
 			break;
 		case CTL_MSG_SYNC_FE:
 			rcv_sync_msg = 1;
 			break;
 		default:
 		        printf("How did I get here?\n");
 		}
 	} else if (event == CTL_HA_EVT_MSG_SENT) {
 		if (param != CTL_HA_STATUS_SUCCESS) {
 			printf("Bad status from ctl_ha_msg_send status %d\n",
 			       param);
 		}
 		return;
 	} else if (event == CTL_HA_EVT_DISCONNECT) {
 		printf("CTL: Got a disconnect from Isc\n");
 		return;
 	} else {
 		printf("ctl_isc_event_handler: Unknown event %d\n", event);
 		return;
 	}
 
 bailout:
 	return;
 }
 
 static void
 ctl_copy_sense_data(union ctl_ha_msg *src, union ctl_io *dest)
 {
 	struct scsi_sense_data *sense;
 
 	sense = &dest->scsiio.sense_data;
 	bcopy(&src->scsi.sense_data, sense, sizeof(*sense));
 	dest->scsiio.scsi_status = src->scsi.scsi_status;
 	dest->scsiio.sense_len = src->scsi.sense_len;
 	dest->io_hdr.status = src->hdr.status;
 }
 
 static void
 ctl_est_ua(struct ctl_lun *lun, uint32_t initidx, ctl_ua_type ua)
 {
 	ctl_ua_type *pu;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 	pu = lun->pending_ua[initidx / CTL_MAX_INIT_PER_PORT];
 	if (pu == NULL)
 		return;
 	pu[initidx % CTL_MAX_INIT_PER_PORT] |= ua;
 }
 
 static void
 ctl_est_ua_all(struct ctl_lun *lun, uint32_t except, ctl_ua_type ua)
 {
 	int i, j;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 	for (i = 0; i < CTL_MAX_PORTS; i++) {
 		if (lun->pending_ua[i] == NULL)
 			continue;
 		for (j = 0; j < CTL_MAX_INIT_PER_PORT; j++) {
 			if (i * CTL_MAX_INIT_PER_PORT + j == except)
 				continue;
 			lun->pending_ua[i][j] |= ua;
 		}
 	}
 }
 
 static void
 ctl_clr_ua(struct ctl_lun *lun, uint32_t initidx, ctl_ua_type ua)
 {
 	ctl_ua_type *pu;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 	pu = lun->pending_ua[initidx / CTL_MAX_INIT_PER_PORT];
 	if (pu == NULL)
 		return;
 	pu[initidx % CTL_MAX_INIT_PER_PORT] &= ~ua;
 }
 
 static void
 ctl_clr_ua_all(struct ctl_lun *lun, uint32_t except, ctl_ua_type ua)
 {
 	int i, j;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 	for (i = 0; i < CTL_MAX_PORTS; i++) {
 		if (lun->pending_ua[i] == NULL)
 			continue;
 		for (j = 0; j < CTL_MAX_INIT_PER_PORT; j++) {
 			if (i * CTL_MAX_INIT_PER_PORT + j == except)
 				continue;
 			lun->pending_ua[i][j] &= ~ua;
 		}
 	}
 }
 
 static int
 ctl_ha_state_sysctl(SYSCTL_HANDLER_ARGS)
 {
 	struct ctl_softc *softc = (struct ctl_softc *)arg1;
 	struct ctl_lun *lun;
 	int error, value;
 
 	if (softc->flags & CTL_FLAG_ACTIVE_SHELF)
 		value = 0;
 	else
 		value = 1;
 
 	error = sysctl_handle_int(oidp, &value, 0, req);
 	if ((error != 0) || (req->newptr == NULL))
 		return (error);
 
 	mtx_lock(&softc->ctl_lock);
 	if (value == 0)
 		softc->flags |= CTL_FLAG_ACTIVE_SHELF;
 	else
 		softc->flags &= ~CTL_FLAG_ACTIVE_SHELF;
 	STAILQ_FOREACH(lun, &softc->lun_list, links) {
 		mtx_lock(&lun->lun_lock);
 		ctl_est_ua_all(lun, -1, CTL_UA_ASYM_ACC_CHANGE);
 		mtx_unlock(&lun->lun_lock);
 	}
 	mtx_unlock(&softc->ctl_lock);
 	return (0);
 }
 
 static int
 ctl_init(void)
 {
 	struct ctl_softc *softc;
 	void *other_pool;
 	struct ctl_port *port;
 	int i, error, retval;
 	//int isc_retval;
 
 	retval = 0;
 	ctl_pause_rtr = 0;
         rcv_sync_msg = 0;
 
 	control_softc = malloc(sizeof(*control_softc), M_DEVBUF,
 			       M_WAITOK | M_ZERO);
 	softc = control_softc;
 
 	softc->dev = make_dev(&ctl_cdevsw, 0, UID_ROOT, GID_OPERATOR, 0600,
 			      "cam/ctl");
 
 	softc->dev->si_drv1 = softc;
 
 	/*
 	 * By default, return a "bad LUN" peripheral qualifier for unknown
 	 * LUNs.  The user can override this default using the tunable or
 	 * sysctl.  See the comment in ctl_inquiry_std() for more details.
 	 */
 	softc->inquiry_pq_no_lun = 1;
 	TUNABLE_INT_FETCH("kern.cam.ctl.inquiry_pq_no_lun",
 			  &softc->inquiry_pq_no_lun);
 	sysctl_ctx_init(&softc->sysctl_ctx);
 	softc->sysctl_tree = SYSCTL_ADD_NODE(&softc->sysctl_ctx,
 		SYSCTL_STATIC_CHILDREN(_kern_cam), OID_AUTO, "ctl",
 		CTLFLAG_RD, 0, "CAM Target Layer");
 
 	if (softc->sysctl_tree == NULL) {
 		printf("%s: unable to allocate sysctl tree\n", __func__);
 		destroy_dev(softc->dev);
 		free(control_softc, M_DEVBUF);
 		control_softc = NULL;
 		return (ENOMEM);
 	}
 
 	SYSCTL_ADD_INT(&softc->sysctl_ctx,
 		       SYSCTL_CHILDREN(softc->sysctl_tree), OID_AUTO,
 		       "inquiry_pq_no_lun", CTLFLAG_RW,
 		       &softc->inquiry_pq_no_lun, 0,
 		       "Report no lun possible for invalid LUNs");
 
 	mtx_init(&softc->ctl_lock, "CTL mutex", NULL, MTX_DEF);
 	softc->io_zone = uma_zcreate("CTL IO", sizeof(union ctl_io),
 	    NULL, NULL, NULL, NULL, UMA_ALIGN_PTR, 0);
 	softc->open_count = 0;
 
 	/*
 	 * Default to actually sending a SYNCHRONIZE CACHE command down to
 	 * the drive.
 	 */
 	softc->flags = CTL_FLAG_REAL_SYNC;
 
 	/*
 	 * In Copan's HA scheme, the "master" and "slave" roles are
 	 * figured out through the slot the controller is in.  Although it
 	 * is an active/active system, someone has to be in charge.
 	 */
 	SYSCTL_ADD_INT(&softc->sysctl_ctx, SYSCTL_CHILDREN(softc->sysctl_tree),
 	    OID_AUTO, "ha_id", CTLFLAG_RDTUN, &softc->ha_id, 0,
 	    "HA head ID (0 - no HA)");
 	if (softc->ha_id == 0) {
 		softc->flags |= CTL_FLAG_ACTIVE_SHELF;
 		softc->is_single = 1;
 		softc->port_offset = 0;
 	} else
 		softc->port_offset = (softc->ha_id - 1) * CTL_MAX_PORTS;
 	softc->persis_offset = softc->port_offset * CTL_MAX_INIT_PER_PORT;
 
 	/*
 	 * XXX KDM need to figure out where we want to get our target ID
 	 * and WWID.  Is it different on each port?
 	 */
 	softc->target.id = 0;
 	softc->target.wwid[0] = 0x12345678;
 	softc->target.wwid[1] = 0x87654321;
 	STAILQ_INIT(&softc->lun_list);
 	STAILQ_INIT(&softc->pending_lun_queue);
 	STAILQ_INIT(&softc->fe_list);
 	STAILQ_INIT(&softc->port_list);
 	STAILQ_INIT(&softc->be_list);
 	ctl_tpc_init(softc);
 
 	if (ctl_pool_create(softc, "othersc", CTL_POOL_ENTRIES_OTHER_SC,
 	                    &other_pool) != 0)
 	{
 		printf("ctl: can't allocate %d entry other SC pool, "
 		       "exiting\n", CTL_POOL_ENTRIES_OTHER_SC);
 		return (ENOMEM);
 	}
 	softc->othersc_pool = other_pool;
 
 	if (worker_threads <= 0)
 		worker_threads = max(1, mp_ncpus / 4);
 	if (worker_threads > CTL_MAX_THREADS)
 		worker_threads = CTL_MAX_THREADS;
 
 	for (i = 0; i < worker_threads; i++) {
 		struct ctl_thread *thr = &softc->threads[i];
 
 		mtx_init(&thr->queue_lock, "CTL queue mutex", NULL, MTX_DEF);
 		thr->ctl_softc = softc;
 		STAILQ_INIT(&thr->incoming_queue);
 		STAILQ_INIT(&thr->rtr_queue);
 		STAILQ_INIT(&thr->done_queue);
 		STAILQ_INIT(&thr->isc_queue);
 
 		error = kproc_kthread_add(ctl_work_thread, thr,
 		    &softc->ctl_proc, &thr->thread, 0, 0, "ctl", "work%d", i);
 		if (error != 0) {
 			printf("error creating CTL work thread!\n");
 			ctl_pool_free(other_pool);
 			return (error);
 		}
 	}
 	error = kproc_kthread_add(ctl_lun_thread, softc,
 	    &softc->ctl_proc, NULL, 0, 0, "ctl", "lun");
 	if (error != 0) {
 		printf("error creating CTL lun thread!\n");
 		ctl_pool_free(other_pool);
 		return (error);
 	}
 	error = kproc_kthread_add(ctl_thresh_thread, softc,
 	    &softc->ctl_proc, NULL, 0, 0, "ctl", "thresh");
 	if (error != 0) {
 		printf("error creating CTL threshold thread!\n");
 		ctl_pool_free(other_pool);
 		return (error);
 	}
 	if (bootverbose)
 		printf("ctl: CAM Target Layer loaded\n");
 
 	/*
 	 * Initialize the ioctl front end.
 	 */
 	ctl_frontend_register(&ioctl_frontend);
 	port = &softc->ioctl_info.port;
 	port->frontend = &ioctl_frontend;
 	sprintf(softc->ioctl_info.port_name, "ioctl");
 	port->port_type = CTL_PORT_IOCTL;
 	port->num_requested_ctl_io = 100;
 	port->port_name = softc->ioctl_info.port_name;
 	port->port_online = ctl_ioctl_online;
 	port->port_offline = ctl_ioctl_offline;
 	port->onoff_arg = &softc->ioctl_info;
 	port->lun_enable = ctl_ioctl_lun_enable;
 	port->lun_disable = ctl_ioctl_lun_disable;
 	port->targ_lun_arg = &softc->ioctl_info;
 	port->fe_datamove = ctl_ioctl_datamove;
 	port->fe_done = ctl_ioctl_done;
 	port->max_targets = 15;
 	port->max_target_id = 15;
 
 	if (ctl_port_register(&softc->ioctl_info.port) != 0) {
 		printf("ctl: ioctl front end registration failed, will "
 		       "continue anyway\n");
 	}
 
 	SYSCTL_ADD_PROC(&softc->sysctl_ctx,SYSCTL_CHILDREN(softc->sysctl_tree),
 	    OID_AUTO, "ha_state", CTLTYPE_INT | CTLFLAG_RWTUN,
 	    softc, 0, ctl_ha_state_sysctl, "I", "HA state for this head");
 
 #ifdef CTL_IO_DELAY
 	if (sizeof(struct callout) > CTL_TIMER_BYTES) {
 		printf("sizeof(struct callout) %zd > CTL_TIMER_BYTES %zd\n",
 		       sizeof(struct callout), CTL_TIMER_BYTES);
 		return (EINVAL);
 	}
 #endif /* CTL_IO_DELAY */
 
 	return (0);
 }
 
 void
 ctl_shutdown(void)
 {
 	struct ctl_softc *softc;
 	struct ctl_lun *lun, *next_lun;
 
 	softc = (struct ctl_softc *)control_softc;
 
 	if (ctl_port_deregister(&softc->ioctl_info.port) != 0)
 		printf("ctl: ioctl front end deregistration failed\n");
 
 	mtx_lock(&softc->ctl_lock);
 
 	/*
 	 * Free up each LUN.
 	 */
 	for (lun = STAILQ_FIRST(&softc->lun_list); lun != NULL; lun = next_lun){
 		next_lun = STAILQ_NEXT(lun, links);
 		ctl_free_lun(lun);
 	}
 
 	mtx_unlock(&softc->ctl_lock);
 
 	ctl_frontend_deregister(&ioctl_frontend);
 
 #if 0
 	ctl_shutdown_thread(softc->work_thread);
 	mtx_destroy(&softc->queue_lock);
 #endif
 
 	ctl_tpc_shutdown(softc);
 	uma_zdestroy(softc->io_zone);
 	mtx_destroy(&softc->ctl_lock);
 
 	destroy_dev(softc->dev);
 
 	sysctl_ctx_free(&softc->sysctl_ctx);
 
 	free(control_softc, M_DEVBUF);
 	control_softc = NULL;
 
 	if (bootverbose)
 		printf("ctl: CAM Target Layer unloaded\n");
 }
 
 static int
 ctl_module_event_handler(module_t mod, int what, void *arg)
 {
 
 	switch (what) {
 	case MOD_LOAD:
 		return (ctl_init());
 	case MOD_UNLOAD:
 		return (EBUSY);
 	default:
 		return (EOPNOTSUPP);
 	}
 }
 
 /*
  * XXX KDM should we do some access checks here?  Bump a reference count to
  * prevent a CTL module from being unloaded while someone has it open?
  */
 static int
 ctl_open(struct cdev *dev, int flags, int fmt, struct thread *td)
 {
 	return (0);
 }
 
 static int
 ctl_close(struct cdev *dev, int flags, int fmt, struct thread *td)
 {
 	return (0);
 }
 
 int
 ctl_port_enable(ctl_port_type port_type)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_port *port;
 
 	if (softc->is_single == 0) {
 		union ctl_ha_msg msg_info;
 		int isc_retval;
 
 #if 0
 		printf("%s: HA mode, synchronizing frontend enable\n",
 		        __func__);
 #endif
 		msg_info.hdr.msg_type = CTL_MSG_SYNC_FE;
 	        if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 		        sizeof(msg_info), 1 )) > CTL_HA_STATUS_SUCCESS) {
 			printf("Sync msg send error retval %d\n", isc_retval);
 		}
 		if (!rcv_sync_msg) {
 			isc_retval=ctl_ha_msg_recv(CTL_HA_CHAN_CTL, &msg_info,
 			        sizeof(msg_info), 1);
 		}
 #if 0
         	printf("CTL:Frontend Enable\n");
 	} else {
 		printf("%s: single mode, skipping frontend synchronization\n",
 		        __func__);
 #endif
 	}
 
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		if (port_type & port->port_type)
 		{
 #if 0
 			printf("port %d\n", port->targ_port);
 #endif
 			ctl_port_online(port);
 		}
 	}
 
 	return (0);
 }
 
 int
 ctl_port_disable(ctl_port_type port_type)
 {
 	struct ctl_softc *softc;
 	struct ctl_port *port;
 
 	softc = control_softc;
 
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		if (port_type & port->port_type)
 			ctl_port_offline(port);
 	}
 
 	return (0);
 }
 
 /*
  * Returns 0 for success, 1 for failure.
  * Currently the only failure mode is if there aren't enough entries
  * allocated.  So, in case of a failure, look at num_entries_dropped,
  * reallocate and try again.
  */
 int
 ctl_port_list(struct ctl_port_entry *entries, int num_entries_alloced,
 	      int *num_entries_filled, int *num_entries_dropped,
 	      ctl_port_type port_type, int no_virtual)
 {
 	struct ctl_softc *softc;
 	struct ctl_port *port;
 	int entries_dropped, entries_filled;
 	int retval;
 	int i;
 
 	softc = control_softc;
 
 	retval = 0;
 	entries_filled = 0;
 	entries_dropped = 0;
 
 	i = 0;
 	mtx_lock(&softc->ctl_lock);
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		struct ctl_port_entry *entry;
 
 		if ((port->port_type & port_type) == 0)
 			continue;
 
 		if ((no_virtual != 0)
 		 && (port->virtual_port != 0))
 			continue;
 
 		if (entries_filled >= num_entries_alloced) {
 			entries_dropped++;
 			continue;
 		}
 		entry = &entries[i];
 
 		entry->port_type = port->port_type;
 		strlcpy(entry->port_name, port->port_name,
 			sizeof(entry->port_name));
 		entry->physical_port = port->physical_port;
 		entry->virtual_port = port->virtual_port;
 		entry->wwnn = port->wwnn;
 		entry->wwpn = port->wwpn;
 
 		i++;
 		entries_filled++;
 	}
 
 	mtx_unlock(&softc->ctl_lock);
 
 	if (entries_dropped > 0)
 		retval = 1;
 
 	*num_entries_dropped = entries_dropped;
 	*num_entries_filled = entries_filled;
 
 	return (retval);
 }
 
 static void
 ctl_ioctl_online(void *arg)
 {
 	struct ctl_ioctl_info *ioctl_info;
 
 	ioctl_info = (struct ctl_ioctl_info *)arg;
 
 	ioctl_info->flags |= CTL_IOCTL_FLAG_ENABLED;
 }
 
 static void
 ctl_ioctl_offline(void *arg)
 {
 	struct ctl_ioctl_info *ioctl_info;
 
 	ioctl_info = (struct ctl_ioctl_info *)arg;
 
 	ioctl_info->flags &= ~CTL_IOCTL_FLAG_ENABLED;
 }
 
 /*
  * Remove an initiator by port number and initiator ID.
  * Returns 0 for success, -1 for failure.
  */
 int
 ctl_remove_initiator(struct ctl_port *port, int iid)
 {
 	struct ctl_softc *softc = control_softc;
 
 	mtx_assert(&softc->ctl_lock, MA_NOTOWNED);
 
 	if (iid > CTL_MAX_INIT_PER_PORT) {
 		printf("%s: initiator ID %u > maximun %u!\n",
 		       __func__, iid, CTL_MAX_INIT_PER_PORT);
 		return (-1);
 	}
 
 	mtx_lock(&softc->ctl_lock);
 	port->wwpn_iid[iid].in_use--;
 	port->wwpn_iid[iid].last_use = time_uptime;
 	mtx_unlock(&softc->ctl_lock);
 
 	return (0);
 }
 
 /*
  * Add an initiator to the initiator map.
  * Returns iid for success, < 0 for failure.
  */
 int
 ctl_add_initiator(struct ctl_port *port, int iid, uint64_t wwpn, char *name)
 {
 	struct ctl_softc *softc = control_softc;
 	time_t best_time;
 	int i, best;
 
 	mtx_assert(&softc->ctl_lock, MA_NOTOWNED);
 
 	if (iid >= CTL_MAX_INIT_PER_PORT) {
 		printf("%s: WWPN %#jx initiator ID %u > maximum %u!\n",
 		       __func__, wwpn, iid, CTL_MAX_INIT_PER_PORT);
 		free(name, M_CTL);
 		return (-1);
 	}
 
 	mtx_lock(&softc->ctl_lock);
 
 	if (iid < 0 && (wwpn != 0 || name != NULL)) {
 		for (i = 0; i < CTL_MAX_INIT_PER_PORT; i++) {
 			if (wwpn != 0 && wwpn == port->wwpn_iid[i].wwpn) {
 				iid = i;
 				break;
 			}
 			if (name != NULL && port->wwpn_iid[i].name != NULL &&
 			    strcmp(name, port->wwpn_iid[i].name) == 0) {
 				iid = i;
 				break;
 			}
 		}
 	}
 
 	if (iid < 0) {
 		for (i = 0; i < CTL_MAX_INIT_PER_PORT; i++) {
 			if (port->wwpn_iid[i].in_use == 0 &&
 			    port->wwpn_iid[i].wwpn == 0 &&
 			    port->wwpn_iid[i].name == NULL) {
 				iid = i;
 				break;
 			}
 		}
 	}
 
 	if (iid < 0) {
 		best = -1;
 		best_time = INT32_MAX;
 		for (i = 0; i < CTL_MAX_INIT_PER_PORT; i++) {
 			if (port->wwpn_iid[i].in_use == 0) {
 				if (port->wwpn_iid[i].last_use < best_time) {
 					best = i;
 					best_time = port->wwpn_iid[i].last_use;
 				}
 			}
 		}
 		iid = best;
 	}
 
 	if (iid < 0) {
 		mtx_unlock(&softc->ctl_lock);
 		free(name, M_CTL);
 		return (-2);
 	}
 
 	if (port->wwpn_iid[iid].in_use > 0 && (wwpn != 0 || name != NULL)) {
 		/*
 		 * This is not an error yet.
 		 */
 		if (wwpn != 0 && wwpn == port->wwpn_iid[iid].wwpn) {
 #if 0
 			printf("%s: port %d iid %u WWPN %#jx arrived"
 			    " again\n", __func__, port->targ_port,
 			    iid, (uintmax_t)wwpn);
 #endif
 			goto take;
 		}
 		if (name != NULL && port->wwpn_iid[iid].name != NULL &&
 		    strcmp(name, port->wwpn_iid[iid].name) == 0) {
 #if 0
 			printf("%s: port %d iid %u name '%s' arrived"
 			    " again\n", __func__, port->targ_port,
 			    iid, name);
 #endif
 			goto take;
 		}
 
 		/*
 		 * This is an error, but what do we do about it?  The
 		 * driver is telling us we have a new WWPN for this
 		 * initiator ID, so we pretty much need to use it.
 		 */
 		printf("%s: port %d iid %u WWPN %#jx '%s' arrived,"
 		    " but WWPN %#jx '%s' is still at that address\n",
 		    __func__, port->targ_port, iid, wwpn, name,
 		    (uintmax_t)port->wwpn_iid[iid].wwpn,
 		    port->wwpn_iid[iid].name);
 
 		/*
 		 * XXX KDM clear have_ca and ua_pending on each LUN for
 		 * this initiator.
 		 */
 	}
 take:
 	free(port->wwpn_iid[iid].name, M_CTL);
 	port->wwpn_iid[iid].name = name;
 	port->wwpn_iid[iid].wwpn = wwpn;
 	port->wwpn_iid[iid].in_use++;
 	mtx_unlock(&softc->ctl_lock);
 
 	return (iid);
 }
 
 static int
 ctl_create_iid(struct ctl_port *port, int iid, uint8_t *buf)
 {
 	int len;
 
 	switch (port->port_type) {
 	case CTL_PORT_FC:
 	{
 		struct scsi_transportid_fcp *id =
 		    (struct scsi_transportid_fcp *)buf;
 		if (port->wwpn_iid[iid].wwpn == 0)
 			return (0);
 		memset(id, 0, sizeof(*id));
 		id->format_protocol = SCSI_PROTO_FC;
 		scsi_u64to8b(port->wwpn_iid[iid].wwpn, id->n_port_name);
 		return (sizeof(*id));
 	}
 	case CTL_PORT_ISCSI:
 	{
 		struct scsi_transportid_iscsi_port *id =
 		    (struct scsi_transportid_iscsi_port *)buf;
 		if (port->wwpn_iid[iid].name == NULL)
 			return (0);
 		memset(id, 0, 256);
 		id->format_protocol = SCSI_TRN_ISCSI_FORMAT_PORT |
 		    SCSI_PROTO_ISCSI;
 		len = strlcpy(id->iscsi_name, port->wwpn_iid[iid].name, 252) + 1;
 		len = roundup2(min(len, 252), 4);
 		scsi_ulto2b(len, id->additional_length);
 		return (sizeof(*id) + len);
 	}
 	case CTL_PORT_SAS:
 	{
 		struct scsi_transportid_sas *id =
 		    (struct scsi_transportid_sas *)buf;
 		if (port->wwpn_iid[iid].wwpn == 0)
 			return (0);
 		memset(id, 0, sizeof(*id));
 		id->format_protocol = SCSI_PROTO_SAS;
 		scsi_u64to8b(port->wwpn_iid[iid].wwpn, id->sas_address);
 		return (sizeof(*id));
 	}
 	default:
 	{
 		struct scsi_transportid_spi *id =
 		    (struct scsi_transportid_spi *)buf;
 		memset(id, 0, sizeof(*id));
 		id->format_protocol = SCSI_PROTO_SPI;
 		scsi_ulto2b(iid, id->scsi_addr);
 		scsi_ulto2b(port->targ_port, id->rel_trgt_port_id);
 		return (sizeof(*id));
 	}
 	}
 }
 
 static int
 ctl_ioctl_lun_enable(void *arg, struct ctl_id targ_id, int lun_id)
 {
 	return (0);
 }
 
 static int
 ctl_ioctl_lun_disable(void *arg, struct ctl_id targ_id, int lun_id)
 {
 	return (0);
 }
 
 /*
  * Data movement routine for the CTL ioctl frontend port.
  */
 static int
 ctl_ioctl_do_datamove(struct ctl_scsiio *ctsio)
 {
 	struct ctl_sg_entry *ext_sglist, *kern_sglist;
 	struct ctl_sg_entry ext_entry, kern_entry;
 	int ext_sglen, ext_sg_entries, kern_sg_entries;
 	int ext_sg_start, ext_offset;
 	int len_to_copy, len_copied;
 	int kern_watermark, ext_watermark;
 	int ext_sglist_malloced;
 	int i, j;
 
 	ext_sglist_malloced = 0;
 	ext_sg_start = 0;
 	ext_offset = 0;
 
 	CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove\n"));
 
 	/*
 	 * If this flag is set, fake the data transfer.
 	 */
 	if (ctsio->io_hdr.flags & CTL_FLAG_NO_DATAMOVE) {
 		ctsio->ext_data_filled = ctsio->ext_data_len;
 		goto bailout;
 	}
 
 	/*
 	 * To simplify things here, if we have a single buffer, stick it in
 	 * a S/G entry and just make it a single entry S/G list.
 	 */
 	if (ctsio->io_hdr.flags & CTL_FLAG_EDPTR_SGLIST) {
 		int len_seen;
 
 		ext_sglen = ctsio->ext_sg_entries * sizeof(*ext_sglist);
 
 		ext_sglist = (struct ctl_sg_entry *)malloc(ext_sglen, M_CTL,
 							   M_WAITOK);
 		ext_sglist_malloced = 1;
 		if (copyin(ctsio->ext_data_ptr, ext_sglist,
 				   ext_sglen) != 0) {
 			ctl_set_internal_failure(ctsio,
 						 /*sks_valid*/ 0,
 						 /*retry_count*/ 0);
 			goto bailout;
 		}
 		ext_sg_entries = ctsio->ext_sg_entries;
 		len_seen = 0;
 		for (i = 0; i < ext_sg_entries; i++) {
 			if ((len_seen + ext_sglist[i].len) >=
 			     ctsio->ext_data_filled) {
 				ext_sg_start = i;
 				ext_offset = ctsio->ext_data_filled - len_seen;
 				break;
 			}
 			len_seen += ext_sglist[i].len;
 		}
 	} else {
 		ext_sglist = &ext_entry;
 		ext_sglist->addr = ctsio->ext_data_ptr;
 		ext_sglist->len = ctsio->ext_data_len;
 		ext_sg_entries = 1;
 		ext_sg_start = 0;
 		ext_offset = ctsio->ext_data_filled;
 	}
 
 	if (ctsio->kern_sg_entries > 0) {
 		kern_sglist = (struct ctl_sg_entry *)ctsio->kern_data_ptr;
 		kern_sg_entries = ctsio->kern_sg_entries;
 	} else {
 		kern_sglist = &kern_entry;
 		kern_sglist->addr = ctsio->kern_data_ptr;
 		kern_sglist->len = ctsio->kern_data_len;
 		kern_sg_entries = 1;
 	}
 
 
 	kern_watermark = 0;
 	ext_watermark = ext_offset;
 	len_copied = 0;
 	for (i = ext_sg_start, j = 0;
 	     i < ext_sg_entries && j < kern_sg_entries;) {
 		uint8_t *ext_ptr, *kern_ptr;
 
 		len_to_copy = MIN(ext_sglist[i].len - ext_watermark,
 				  kern_sglist[j].len - kern_watermark);
 
 		ext_ptr = (uint8_t *)ext_sglist[i].addr;
 		ext_ptr = ext_ptr + ext_watermark;
 		if (ctsio->io_hdr.flags & CTL_FLAG_BUS_ADDR) {
 			/*
 			 * XXX KDM fix this!
 			 */
 			panic("need to implement bus address support");
 #if 0
 			kern_ptr = bus_to_virt(kern_sglist[j].addr);
 #endif
 		} else
 			kern_ptr = (uint8_t *)kern_sglist[j].addr;
 		kern_ptr = kern_ptr + kern_watermark;
 
 		kern_watermark += len_to_copy;
 		ext_watermark += len_to_copy;
 
 		if ((ctsio->io_hdr.flags & CTL_FLAG_DATA_MASK) ==
 		     CTL_FLAG_DATA_IN) {
 			CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove: copying %d "
 					 "bytes to user\n", len_to_copy));
 			CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove: from %p "
 					 "to %p\n", kern_ptr, ext_ptr));
 			if (copyout(kern_ptr, ext_ptr, len_to_copy) != 0) {
 				ctl_set_internal_failure(ctsio,
 							 /*sks_valid*/ 0,
 							 /*retry_count*/ 0);
 				goto bailout;
 			}
 		} else {
 			CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove: copying %d "
 					 "bytes from user\n", len_to_copy));
 			CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove: from %p "
 					 "to %p\n", ext_ptr, kern_ptr));
 			if (copyin(ext_ptr, kern_ptr, len_to_copy)!= 0){
 				ctl_set_internal_failure(ctsio,
 							 /*sks_valid*/ 0,
 							 /*retry_count*/0);
 				goto bailout;
 			}
 		}
 
 		len_copied += len_to_copy;
 
 		if (ext_sglist[i].len == ext_watermark) {
 			i++;
 			ext_watermark = 0;
 		}
 
 		if (kern_sglist[j].len == kern_watermark) {
 			j++;
 			kern_watermark = 0;
 		}
 	}
 
 	ctsio->ext_data_filled += len_copied;
 
 	CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove: ext_sg_entries: %d, "
 			 "kern_sg_entries: %d\n", ext_sg_entries,
 			 kern_sg_entries));
 	CTL_DEBUG_PRINT(("ctl_ioctl_do_datamove: ext_data_len = %d, "
 			 "kern_data_len = %d\n", ctsio->ext_data_len,
 			 ctsio->kern_data_len));
 
 
 	/* XXX KDM set residual?? */
 bailout:
 
 	if (ext_sglist_malloced != 0)
 		free(ext_sglist, M_CTL);
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 /*
  * Serialize a command that went down the "wrong" side, and so was sent to
  * this controller for execution.  The logic is a little different than the
  * standard case in ctl_scsiio_precheck().  Errors in this case need to get
  * sent back to the other side, but in the success case, we execute the
  * command on this side (XFER mode) or tell the other side to execute it
  * (SER_ONLY mode).
  */
 static int
 ctl_serialize_other_sc_cmd(struct ctl_scsiio *ctsio)
 {
 	struct ctl_softc *softc;
 	union ctl_ha_msg msg_info;
 	struct ctl_lun *lun;
 	int retval = 0;
 	uint32_t targ_lun;
 
 	softc = control_softc;
 
 	targ_lun = ctsio->io_hdr.nexus.targ_mapped_lun;
 	lun = softc->ctl_luns[targ_lun];
 	if (lun==NULL)
 	{
 		/*
 		 * Why isn't LUN defined? The other side wouldn't
 		 * send a cmd if the LUN is undefined.
 		 */
 		printf("%s: Bad JUJU!, LUN is NULL!\n", __func__);
 
 		/* "Logical unit not supported" */
 		ctl_set_sense_data(&msg_info.scsi.sense_data,
 				   lun,
 				   /*sense_format*/SSD_TYPE_NONE,
 				   /*current_error*/ 1,
 				   /*sense_key*/ SSD_KEY_ILLEGAL_REQUEST,
 				   /*asc*/ 0x25,
 				   /*ascq*/ 0x00,
 				   SSD_ELEM_NONE);
 
 		msg_info.scsi.sense_len = SSD_FULL_SIZE;
 		msg_info.scsi.scsi_status = SCSI_STATUS_CHECK_COND;
 		msg_info.hdr.status = CTL_SCSI_ERROR | CTL_AUTOSENSE;
 		msg_info.hdr.original_sc = ctsio->io_hdr.original_sc;
 		msg_info.hdr.serializing_sc = NULL;
 		msg_info.hdr.msg_type = CTL_MSG_BAD_JUJU;
 	        if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 				sizeof(msg_info), 0 ) > CTL_HA_STATUS_SUCCESS) {
 		}
 		return(1);
 
 	}
 
 	mtx_lock(&lun->lun_lock);
     	TAILQ_INSERT_TAIL(&lun->ooa_queue, &ctsio->io_hdr, ooa_links);
 
 	switch (ctl_check_ooa(lun, (union ctl_io *)ctsio,
 		(union ctl_io *)TAILQ_PREV(&ctsio->io_hdr, ctl_ooaq,
 		 ooa_links))) {
 	case CTL_ACTION_BLOCK:
 		ctsio->io_hdr.flags |= CTL_FLAG_BLOCKED;
 		TAILQ_INSERT_TAIL(&lun->blocked_queue, &ctsio->io_hdr,
 				  blocked_links);
 		break;
 	case CTL_ACTION_PASS:
 	case CTL_ACTION_SKIP:
 		if (softc->ha_mode == CTL_HA_MODE_XFER) {
 			ctsio->io_hdr.flags |= CTL_FLAG_IS_WAS_ON_RTR;
 			ctl_enqueue_rtr((union ctl_io *)ctsio);
 		} else {
 
 			/* send msg back to other side */
 			msg_info.hdr.original_sc = ctsio->io_hdr.original_sc;
 			msg_info.hdr.serializing_sc = (union ctl_io *)ctsio;
 			msg_info.hdr.msg_type = CTL_MSG_R2R;
 #if 0
 			printf("2. pOrig %x\n", (int)msg_info.hdr.original_sc);
 #endif
 		        if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 			    sizeof(msg_info), 0 ) > CTL_HA_STATUS_SUCCESS) {
 			}
 		}
 		break;
 	case CTL_ACTION_OVERLAP:
 		/* OVERLAPPED COMMANDS ATTEMPTED */
 		ctl_set_sense_data(&msg_info.scsi.sense_data,
 				   lun,
 				   /*sense_format*/SSD_TYPE_NONE,
 				   /*current_error*/ 1,
 				   /*sense_key*/ SSD_KEY_ILLEGAL_REQUEST,
 				   /*asc*/ 0x4E,
 				   /*ascq*/ 0x00,
 				   SSD_ELEM_NONE);
 
 		msg_info.scsi.sense_len = SSD_FULL_SIZE;
 		msg_info.scsi.scsi_status = SCSI_STATUS_CHECK_COND;
 		msg_info.hdr.status = CTL_SCSI_ERROR | CTL_AUTOSENSE;
 		msg_info.hdr.original_sc = ctsio->io_hdr.original_sc;
 		msg_info.hdr.serializing_sc = NULL;
 		msg_info.hdr.msg_type = CTL_MSG_BAD_JUJU;
 #if 0
 		printf("BAD JUJU:Major Bummer Overlap\n");
 #endif
 		TAILQ_REMOVE(&lun->ooa_queue, &ctsio->io_hdr, ooa_links);
 		retval = 1;
 		if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 		    sizeof(msg_info), 0 ) > CTL_HA_STATUS_SUCCESS) {
 		}
 		break;
 	case CTL_ACTION_OVERLAP_TAG:
 		/* TAGGED OVERLAPPED COMMANDS (NN = QUEUE TAG) */
 		ctl_set_sense_data(&msg_info.scsi.sense_data,
 				   lun,
 				   /*sense_format*/SSD_TYPE_NONE,
 				   /*current_error*/ 1,
 				   /*sense_key*/ SSD_KEY_ILLEGAL_REQUEST,
 				   /*asc*/ 0x4D,
 				   /*ascq*/ ctsio->tag_num & 0xff,
 				   SSD_ELEM_NONE);
 
 		msg_info.scsi.sense_len = SSD_FULL_SIZE;
 		msg_info.scsi.scsi_status = SCSI_STATUS_CHECK_COND;
 		msg_info.hdr.status = CTL_SCSI_ERROR | CTL_AUTOSENSE;
 		msg_info.hdr.original_sc = ctsio->io_hdr.original_sc;
 		msg_info.hdr.serializing_sc = NULL;
 		msg_info.hdr.msg_type = CTL_MSG_BAD_JUJU;
 #if 0
 		printf("BAD JUJU:Major Bummer Overlap Tag\n");
 #endif
 		TAILQ_REMOVE(&lun->ooa_queue, &ctsio->io_hdr, ooa_links);
 		retval = 1;
 		if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 		    sizeof(msg_info), 0 ) > CTL_HA_STATUS_SUCCESS) {
 		}
 		break;
 	case CTL_ACTION_ERROR:
 	default:
 		/* "Internal target failure" */
 		ctl_set_sense_data(&msg_info.scsi.sense_data,
 				   lun,
 				   /*sense_format*/SSD_TYPE_NONE,
 				   /*current_error*/ 1,
 				   /*sense_key*/ SSD_KEY_HARDWARE_ERROR,
 				   /*asc*/ 0x44,
 				   /*ascq*/ 0x00,
 				   SSD_ELEM_NONE);
 
 		msg_info.scsi.sense_len = SSD_FULL_SIZE;
 		msg_info.scsi.scsi_status = SCSI_STATUS_CHECK_COND;
 		msg_info.hdr.status = CTL_SCSI_ERROR | CTL_AUTOSENSE;
 		msg_info.hdr.original_sc = ctsio->io_hdr.original_sc;
 		msg_info.hdr.serializing_sc = NULL;
 		msg_info.hdr.msg_type = CTL_MSG_BAD_JUJU;
 #if 0
 		printf("BAD JUJU:Major Bummer HW Error\n");
 #endif
 		TAILQ_REMOVE(&lun->ooa_queue, &ctsio->io_hdr, ooa_links);
 		retval = 1;
 		if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_info,
 		    sizeof(msg_info), 0 ) > CTL_HA_STATUS_SUCCESS) {
 		}
 		break;
 	}
 	mtx_unlock(&lun->lun_lock);
 	return (retval);
 }
 
 static int
 ctl_ioctl_submit_wait(union ctl_io *io)
 {
 	struct ctl_fe_ioctl_params params;
 	ctl_fe_ioctl_state last_state;
 	int done, retval;
 
 	retval = 0;
 
 	bzero(&params, sizeof(params));
 
 	mtx_init(&params.ioctl_mtx, "ctliocmtx", NULL, MTX_DEF);
 	cv_init(&params.sem, "ctlioccv");
 	params.state = CTL_IOCTL_INPROG;
 	last_state = params.state;
 
 	io->io_hdr.ctl_private[CTL_PRIV_FRONTEND].ptr = &params;
 
 	CTL_DEBUG_PRINT(("ctl_ioctl_submit_wait\n"));
 
 	/* This shouldn't happen */
 	if ((retval = ctl_queue(io)) != CTL_RETVAL_COMPLETE)
 		return (retval);
 
 	done = 0;
 
 	do {
 		mtx_lock(&params.ioctl_mtx);
 		/*
 		 * Check the state here, and don't sleep if the state has
 		 * already changed (i.e. wakeup has already occured, but we
 		 * weren't waiting yet).
 		 */
 		if (params.state == last_state) {
 			/* XXX KDM cv_wait_sig instead? */
 			cv_wait(&params.sem, &params.ioctl_mtx);
 		}
 		last_state = params.state;
 
 		switch (params.state) {
 		case CTL_IOCTL_INPROG:
 			/* Why did we wake up? */
 			/* XXX KDM error here? */
 			mtx_unlock(&params.ioctl_mtx);
 			break;
 		case CTL_IOCTL_DATAMOVE:
 			CTL_DEBUG_PRINT(("got CTL_IOCTL_DATAMOVE\n"));
 
 			/*
 			 * change last_state back to INPROG to avoid
 			 * deadlock on subsequent data moves.
 			 */
 			params.state = last_state = CTL_IOCTL_INPROG;
 
 			mtx_unlock(&params.ioctl_mtx);
 			ctl_ioctl_do_datamove(&io->scsiio);
 			/*
 			 * Note that in some cases, most notably writes,
 			 * this will queue the I/O and call us back later.
 			 * In other cases, generally reads, this routine
 			 * will immediately call back and wake us up,
 			 * probably using our own context.
 			 */
 			io->scsiio.be_move_done(io);
 			break;
 		case CTL_IOCTL_DONE:
 			mtx_unlock(&params.ioctl_mtx);
 			CTL_DEBUG_PRINT(("got CTL_IOCTL_DONE\n"));
 			done = 1;
 			break;
 		default:
 			mtx_unlock(&params.ioctl_mtx);
 			/* XXX KDM error here? */
 			break;
 		}
 	} while (done == 0);
 
 	mtx_destroy(&params.ioctl_mtx);
 	cv_destroy(&params.sem);
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static void
 ctl_ioctl_datamove(union ctl_io *io)
 {
 	struct ctl_fe_ioctl_params *params;
 
 	params = (struct ctl_fe_ioctl_params *)
 		io->io_hdr.ctl_private[CTL_PRIV_FRONTEND].ptr;
 
 	mtx_lock(&params->ioctl_mtx);
 	params->state = CTL_IOCTL_DATAMOVE;
 	cv_broadcast(&params->sem);
 	mtx_unlock(&params->ioctl_mtx);
 }
 
 static void
 ctl_ioctl_done(union ctl_io *io)
 {
 	struct ctl_fe_ioctl_params *params;
 
 	params = (struct ctl_fe_ioctl_params *)
 		io->io_hdr.ctl_private[CTL_PRIV_FRONTEND].ptr;
 
 	mtx_lock(&params->ioctl_mtx);
 	params->state = CTL_IOCTL_DONE;
 	cv_broadcast(&params->sem);
 	mtx_unlock(&params->ioctl_mtx);
 }
 
 static void
 ctl_ioctl_hard_startstop_callback(void *arg, struct cfi_metatask *metatask)
 {
 	struct ctl_fe_ioctl_startstop_info *sd_info;
 
 	sd_info = (struct ctl_fe_ioctl_startstop_info *)arg;
 
 	sd_info->hs_info.status = metatask->status;
 	sd_info->hs_info.total_luns = metatask->taskinfo.startstop.total_luns;
 	sd_info->hs_info.luns_complete =
 		metatask->taskinfo.startstop.luns_complete;
 	sd_info->hs_info.luns_failed = metatask->taskinfo.startstop.luns_failed;
 
 	cv_broadcast(&sd_info->sem);
 }
 
 static void
 ctl_ioctl_bbrread_callback(void *arg, struct cfi_metatask *metatask)
 {
 	struct ctl_fe_ioctl_bbrread_info *fe_bbr_info;
 
 	fe_bbr_info = (struct ctl_fe_ioctl_bbrread_info *)arg;
 
 	mtx_lock(fe_bbr_info->lock);
 	fe_bbr_info->bbr_info->status = metatask->status;
 	fe_bbr_info->bbr_info->bbr_status = metatask->taskinfo.bbrread.status;
 	fe_bbr_info->wakeup_done = 1;
 	mtx_unlock(fe_bbr_info->lock);
 
 	cv_broadcast(&fe_bbr_info->sem);
 }
 
 /*
  * Returns 0 for success, errno for failure.
  */
 static int
 ctl_ioctl_fill_ooa(struct ctl_lun *lun, uint32_t *cur_fill_num,
 		   struct ctl_ooa *ooa_hdr, struct ctl_ooa_entry *kern_entries)
 {
 	union ctl_io *io;
 	int retval;
 
 	retval = 0;
 
 	mtx_lock(&lun->lun_lock);
 	for (io = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue); (io != NULL);
 	     (*cur_fill_num)++, io = (union ctl_io *)TAILQ_NEXT(&io->io_hdr,
 	     ooa_links)) {
 		struct ctl_ooa_entry *entry;
 
 		/*
 		 * If we've got more than we can fit, just count the
 		 * remaining entries.
 		 */
 		if (*cur_fill_num >= ooa_hdr->alloc_num)
 			continue;
 
 		entry = &kern_entries[*cur_fill_num];
 
 		entry->tag_num = io->scsiio.tag_num;
 		entry->lun_num = lun->lun;
 #ifdef CTL_TIME_IO
 		entry->start_bt = io->io_hdr.start_bt;
 #endif
 		bcopy(io->scsiio.cdb, entry->cdb, io->scsiio.cdb_len);
 		entry->cdb_len = io->scsiio.cdb_len;
 		if (io->io_hdr.flags & CTL_FLAG_BLOCKED)
 			entry->cmd_flags |= CTL_OOACMD_FLAG_BLOCKED;
 
 		if (io->io_hdr.flags & CTL_FLAG_DMA_INPROG)
 			entry->cmd_flags |= CTL_OOACMD_FLAG_DMA;
 
 		if (io->io_hdr.flags & CTL_FLAG_ABORT)
 			entry->cmd_flags |= CTL_OOACMD_FLAG_ABORT;
 
 		if (io->io_hdr.flags & CTL_FLAG_IS_WAS_ON_RTR)
 			entry->cmd_flags |= CTL_OOACMD_FLAG_RTR;
 
 		if (io->io_hdr.flags & CTL_FLAG_DMA_QUEUED)
 			entry->cmd_flags |= CTL_OOACMD_FLAG_DMA_QUEUED;
 	}
 	mtx_unlock(&lun->lun_lock);
 
 	return (retval);
 }
 
 static void *
 ctl_copyin_alloc(void *user_addr, int len, char *error_str,
 		 size_t error_str_len)
 {
 	void *kptr;
 
 	kptr = malloc(len, M_CTL, M_WAITOK | M_ZERO);
 
 	if (copyin(user_addr, kptr, len) != 0) {
 		snprintf(error_str, error_str_len, "Error copying %d bytes "
 			 "from user address %p to kernel address %p", len,
 			 user_addr, kptr);
 		free(kptr, M_CTL);
 		return (NULL);
 	}
 
 	return (kptr);
 }
 
 static void
 ctl_free_args(int num_args, struct ctl_be_arg *args)
 {
 	int i;
 
 	if (args == NULL)
 		return;
 
 	for (i = 0; i < num_args; i++) {
 		free(args[i].kname, M_CTL);
 		free(args[i].kvalue, M_CTL);
 	}
 
 	free(args, M_CTL);
 }
 
 static struct ctl_be_arg *
 ctl_copyin_args(int num_args, struct ctl_be_arg *uargs,
 		char *error_str, size_t error_str_len)
 {
 	struct ctl_be_arg *args;
 	int i;
 
 	args = ctl_copyin_alloc(uargs, num_args * sizeof(*args),
 				error_str, error_str_len);
 
 	if (args == NULL)
 		goto bailout;
 
 	for (i = 0; i < num_args; i++) {
 		args[i].kname = NULL;
 		args[i].kvalue = NULL;
 	}
 
 	for (i = 0; i < num_args; i++) {
 		uint8_t *tmpptr;
 
 		args[i].kname = ctl_copyin_alloc(args[i].name,
 			args[i].namelen, error_str, error_str_len);
 		if (args[i].kname == NULL)
 			goto bailout;
 
 		if (args[i].kname[args[i].namelen - 1] != '\0') {
 			snprintf(error_str, error_str_len, "Argument %d "
 				 "name is not NUL-terminated", i);
 			goto bailout;
 		}
 
 		if (args[i].flags & CTL_BEARG_RD) {
 			tmpptr = ctl_copyin_alloc(args[i].value,
 				args[i].vallen, error_str, error_str_len);
 			if (tmpptr == NULL)
 				goto bailout;
 			if ((args[i].flags & CTL_BEARG_ASCII)
 			 && (tmpptr[args[i].vallen - 1] != '\0')) {
 				snprintf(error_str, error_str_len, "Argument "
 				    "%d value is not NUL-terminated", i);
 				goto bailout;
 			}
 			args[i].kvalue = tmpptr;
 		} else {
 			args[i].kvalue = malloc(args[i].vallen,
 			    M_CTL, M_WAITOK | M_ZERO);
 		}
 	}
 
 	return (args);
 bailout:
 
 	ctl_free_args(num_args, args);
 
 	return (NULL);
 }
 
 static void
 ctl_copyout_args(int num_args, struct ctl_be_arg *args)
 {
 	int i;
 
 	for (i = 0; i < num_args; i++) {
 		if (args[i].flags & CTL_BEARG_WR)
 			copyout(args[i].kvalue, args[i].value, args[i].vallen);
 	}
 }
 
 /*
  * Escape characters that are illegal or not recommended in XML.
  */
 int
 ctl_sbuf_printf_esc(struct sbuf *sb, char *str, int size)
 {
 	char *end = str + size;
 	int retval;
 
 	retval = 0;
 
 	for (; *str && str < end; str++) {
 		switch (*str) {
 		case '&':
 			retval = sbuf_printf(sb, "&amp;");
 			break;
 		case '>':
 			retval = sbuf_printf(sb, "&gt;");
 			break;
 		case '<':
 			retval = sbuf_printf(sb, "&lt;");
 			break;
 		default:
 			retval = sbuf_putc(sb, *str);
 			break;
 		}
 
 		if (retval != 0)
 			break;
 
 	}
 
 	return (retval);
 }
 
 static void
 ctl_id_sbuf(struct ctl_devid *id, struct sbuf *sb)
 {
 	struct scsi_vpd_id_descriptor *desc;
 	int i;
 
 	if (id == NULL || id->len < 4)
 		return;
 	desc = (struct scsi_vpd_id_descriptor *)id->data;
 	switch (desc->id_type & SVPD_ID_TYPE_MASK) {
 	case SVPD_ID_TYPE_T10:
 		sbuf_printf(sb, "t10.");
 		break;
 	case SVPD_ID_TYPE_EUI64:
 		sbuf_printf(sb, "eui.");
 		break;
 	case SVPD_ID_TYPE_NAA:
 		sbuf_printf(sb, "naa.");
 		break;
 	case SVPD_ID_TYPE_SCSI_NAME:
 		break;
 	}
 	switch (desc->proto_codeset & SVPD_ID_CODESET_MASK) {
 	case SVPD_ID_CODESET_BINARY:
 		for (i = 0; i < desc->length; i++)
 			sbuf_printf(sb, "%02x", desc->identifier[i]);
 		break;
 	case SVPD_ID_CODESET_ASCII:
 		sbuf_printf(sb, "%.*s", (int)desc->length,
 		    (char *)desc->identifier);
 		break;
 	case SVPD_ID_CODESET_UTF8:
 		sbuf_printf(sb, "%s", (char *)desc->identifier);
 		break;
 	}
 }
 
 static int
 ctl_ioctl(struct cdev *dev, u_long cmd, caddr_t addr, int flag,
 	  struct thread *td)
 {
 	struct ctl_softc *softc;
 	int retval;
 
 	softc = control_softc;
 
 	retval = 0;
 
 	switch (cmd) {
 	case CTL_IO: {
 		union ctl_io *io;
 		void *pool_tmp;
 
 		/*
 		 * If we haven't been "enabled", don't allow any SCSI I/O
 		 * to this FETD.
 		 */
 		if ((softc->ioctl_info.flags & CTL_IOCTL_FLAG_ENABLED) == 0) {
 			retval = EPERM;
 			break;
 		}
 
 		io = ctl_alloc_io(softc->ioctl_info.port.ctl_pool_ref);
 
 		/*
 		 * Need to save the pool reference so it doesn't get
 		 * spammed by the user's ctl_io.
 		 */
 		pool_tmp = io->io_hdr.pool;
 		memcpy(io, (void *)addr, sizeof(*io));
 		io->io_hdr.pool = pool_tmp;
 
 		/*
 		 * No status yet, so make sure the status is set properly.
 		 */
 		io->io_hdr.status = CTL_STATUS_NONE;
 
 		/*
 		 * The user sets the initiator ID, target and LUN IDs.
 		 */
 		io->io_hdr.nexus.targ_port = softc->ioctl_info.port.targ_port;
 		io->io_hdr.flags |= CTL_FLAG_USER_REQ;
 		if ((io->io_hdr.io_type == CTL_IO_SCSI)
 		 && (io->scsiio.tag_type != CTL_TAG_UNTAGGED))
 			io->scsiio.tag_num = softc->ioctl_info.cur_tag_num++;
 
 		retval = ctl_ioctl_submit_wait(io);
 
 		if (retval != 0) {
 			ctl_free_io(io);
 			break;
 		}
 
 		memcpy((void *)addr, io, sizeof(*io));
 
 		/* return this to our pool */
 		ctl_free_io(io);
 
 		break;
 	}
 	case CTL_ENABLE_PORT:
 	case CTL_DISABLE_PORT:
 	case CTL_SET_PORT_WWNS: {
 		struct ctl_port *port;
 		struct ctl_port_entry *entry;
 
 		entry = (struct ctl_port_entry *)addr;
 		
 		mtx_lock(&softc->ctl_lock);
 		STAILQ_FOREACH(port, &softc->port_list, links) {
 			int action, done;
 
 			action = 0;
 			done = 0;
 
 			if ((entry->port_type == CTL_PORT_NONE)
 			 && (entry->targ_port == port->targ_port)) {
 				/*
 				 * If the user only wants to enable or
 				 * disable or set WWNs on a specific port,
 				 * do the operation and we're done.
 				 */
 				action = 1;
 				done = 1;
 			} else if (entry->port_type & port->port_type) {
 				/*
 				 * Compare the user's type mask with the
 				 * particular frontend type to see if we
 				 * have a match.
 				 */
 				action = 1;
 				done = 0;
 
 				/*
 				 * Make sure the user isn't trying to set
 				 * WWNs on multiple ports at the same time.
 				 */
 				if (cmd == CTL_SET_PORT_WWNS) {
 					printf("%s: Can't set WWNs on "
 					       "multiple ports\n", __func__);
 					retval = EINVAL;
 					break;
 				}
 			}
 			if (action != 0) {
 				/*
 				 * XXX KDM we have to drop the lock here,
 				 * because the online/offline operations
 				 * can potentially block.  We need to
 				 * reference count the frontends so they
 				 * can't go away,
 				 */
 				mtx_unlock(&softc->ctl_lock);
 
 				if (cmd == CTL_ENABLE_PORT) {
 					struct ctl_lun *lun;
 
 					STAILQ_FOREACH(lun, &softc->lun_list,
 						       links) {
 						port->lun_enable(port->targ_lun_arg,
 						    lun->target,
 						    lun->lun);
 					}
 
 					ctl_port_online(port);
 				} else if (cmd == CTL_DISABLE_PORT) {
 					struct ctl_lun *lun;
 
 					ctl_port_offline(port);
 
 					STAILQ_FOREACH(lun, &softc->lun_list,
 						       links) {
 						port->lun_disable(
 						    port->targ_lun_arg,
 						    lun->target,
 						    lun->lun);
 					}
 				}
 
 				mtx_lock(&softc->ctl_lock);
 
 				if (cmd == CTL_SET_PORT_WWNS)
 					ctl_port_set_wwns(port,
 					    (entry->flags & CTL_PORT_WWNN_VALID) ?
 					    1 : 0, entry->wwnn,
 					    (entry->flags & CTL_PORT_WWPN_VALID) ?
 					    1 : 0, entry->wwpn);
 			}
 			if (done != 0)
 				break;
 		}
 		mtx_unlock(&softc->ctl_lock);
 		break;
 	}
 	case CTL_GET_PORT_LIST: {
 		struct ctl_port *port;
 		struct ctl_port_list *list;
 		int i;
 
 		list = (struct ctl_port_list *)addr;
 
 		if (list->alloc_len != (list->alloc_num *
 		    sizeof(struct ctl_port_entry))) {
 			printf("%s: CTL_GET_PORT_LIST: alloc_len %u != "
 			       "alloc_num %u * sizeof(struct ctl_port_entry) "
 			       "%zu\n", __func__, list->alloc_len,
 			       list->alloc_num, sizeof(struct ctl_port_entry));
 			retval = EINVAL;
 			break;
 		}
 		list->fill_len = 0;
 		list->fill_num = 0;
 		list->dropped_num = 0;
 		i = 0;
 		mtx_lock(&softc->ctl_lock);
 		STAILQ_FOREACH(port, &softc->port_list, links) {
 			struct ctl_port_entry entry, *list_entry;
 
 			if (list->fill_num >= list->alloc_num) {
 				list->dropped_num++;
 				continue;
 			}
 
 			entry.port_type = port->port_type;
 			strlcpy(entry.port_name, port->port_name,
 				sizeof(entry.port_name));
 			entry.targ_port = port->targ_port;
 			entry.physical_port = port->physical_port;
 			entry.virtual_port = port->virtual_port;
 			entry.wwnn = port->wwnn;
 			entry.wwpn = port->wwpn;
 			if (port->status & CTL_PORT_STATUS_ONLINE)
 				entry.online = 1;
 			else
 				entry.online = 0;
 
 			list_entry = &list->entries[i];
 
 			retval = copyout(&entry, list_entry, sizeof(entry));
 			if (retval != 0) {
 				printf("%s: CTL_GET_PORT_LIST: copyout "
 				       "returned %d\n", __func__, retval);
 				break;
 			}
 			i++;
 			list->fill_num++;
 			list->fill_len += sizeof(entry);
 		}
 		mtx_unlock(&softc->ctl_lock);
 
 		/*
 		 * If this is non-zero, we had a copyout fault, so there's
 		 * probably no point in attempting to set the status inside
 		 * the structure.
 		 */
 		if (retval != 0)
 			break;
 
 		if (list->dropped_num > 0)
 			list->status = CTL_PORT_LIST_NEED_MORE_SPACE;
 		else
 			list->status = CTL_PORT_LIST_OK;
 		break;
 	}
 	case CTL_DUMP_OOA: {
 		struct ctl_lun *lun;
 		union ctl_io *io;
 		char printbuf[128];
 		struct sbuf sb;
 
 		mtx_lock(&softc->ctl_lock);
 		printf("Dumping OOA queues:\n");
 		STAILQ_FOREACH(lun, &softc->lun_list, links) {
 			mtx_lock(&lun->lun_lock);
 			for (io = (union ctl_io *)TAILQ_FIRST(
 			     &lun->ooa_queue); io != NULL;
 			     io = (union ctl_io *)TAILQ_NEXT(&io->io_hdr,
 			     ooa_links)) {
 				sbuf_new(&sb, printbuf, sizeof(printbuf),
 					 SBUF_FIXEDLEN);
 				sbuf_printf(&sb, "LUN %jd tag 0x%04x%s%s%s%s: ",
 					    (intmax_t)lun->lun,
 					    io->scsiio.tag_num,
 					    (io->io_hdr.flags &
 					    CTL_FLAG_BLOCKED) ? "" : " BLOCKED",
 					    (io->io_hdr.flags &
 					    CTL_FLAG_DMA_INPROG) ? " DMA" : "",
 					    (io->io_hdr.flags &
 					    CTL_FLAG_ABORT) ? " ABORT" : "",
 			                    (io->io_hdr.flags &
 		                        CTL_FLAG_IS_WAS_ON_RTR) ? " RTR" : "");
 				ctl_scsi_command_string(&io->scsiio, NULL, &sb);
 				sbuf_finish(&sb);
 				printf("%s\n", sbuf_data(&sb));
 			}
 			mtx_unlock(&lun->lun_lock);
 		}
 		printf("OOA queues dump done\n");
 		mtx_unlock(&softc->ctl_lock);
 		break;
 	}
 	case CTL_GET_OOA: {
 		struct ctl_lun *lun;
 		struct ctl_ooa *ooa_hdr;
 		struct ctl_ooa_entry *entries;
 		uint32_t cur_fill_num;
 
 		ooa_hdr = (struct ctl_ooa *)addr;
 
 		if ((ooa_hdr->alloc_len == 0)
 		 || (ooa_hdr->alloc_num == 0)) {
 			printf("%s: CTL_GET_OOA: alloc len %u and alloc num %u "
 			       "must be non-zero\n", __func__,
 			       ooa_hdr->alloc_len, ooa_hdr->alloc_num);
 			retval = EINVAL;
 			break;
 		}
 
 		if (ooa_hdr->alloc_len != (ooa_hdr->alloc_num *
 		    sizeof(struct ctl_ooa_entry))) {
 			printf("%s: CTL_GET_OOA: alloc len %u must be alloc "
 			       "num %d * sizeof(struct ctl_ooa_entry) %zd\n",
 			       __func__, ooa_hdr->alloc_len,
 			       ooa_hdr->alloc_num,sizeof(struct ctl_ooa_entry));
 			retval = EINVAL;
 			break;
 		}
 
 		entries = malloc(ooa_hdr->alloc_len, M_CTL, M_WAITOK | M_ZERO);
 		if (entries == NULL) {
 			printf("%s: could not allocate %d bytes for OOA "
 			       "dump\n", __func__, ooa_hdr->alloc_len);
 			retval = ENOMEM;
 			break;
 		}
 
 		mtx_lock(&softc->ctl_lock);
 		if (((ooa_hdr->flags & CTL_OOA_FLAG_ALL_LUNS) == 0)
 		 && ((ooa_hdr->lun_num >= CTL_MAX_LUNS)
 		  || (softc->ctl_luns[ooa_hdr->lun_num] == NULL))) {
 			mtx_unlock(&softc->ctl_lock);
 			free(entries, M_CTL);
 			printf("%s: CTL_GET_OOA: invalid LUN %ju\n",
 			       __func__, (uintmax_t)ooa_hdr->lun_num);
 			retval = EINVAL;
 			break;
 		}
 
 		cur_fill_num = 0;
 
 		if (ooa_hdr->flags & CTL_OOA_FLAG_ALL_LUNS) {
 			STAILQ_FOREACH(lun, &softc->lun_list, links) {
 				retval = ctl_ioctl_fill_ooa(lun, &cur_fill_num,
 					ooa_hdr, entries);
 				if (retval != 0)
 					break;
 			}
 			if (retval != 0) {
 				mtx_unlock(&softc->ctl_lock);
 				free(entries, M_CTL);
 				break;
 			}
 		} else {
 			lun = softc->ctl_luns[ooa_hdr->lun_num];
 
 			retval = ctl_ioctl_fill_ooa(lun, &cur_fill_num,ooa_hdr,
 						    entries);
 		}
 		mtx_unlock(&softc->ctl_lock);
 
 		ooa_hdr->fill_num = min(cur_fill_num, ooa_hdr->alloc_num);
 		ooa_hdr->fill_len = ooa_hdr->fill_num *
 			sizeof(struct ctl_ooa_entry);
 		retval = copyout(entries, ooa_hdr->entries, ooa_hdr->fill_len);
 		if (retval != 0) {
 			printf("%s: error copying out %d bytes for OOA dump\n", 
 			       __func__, ooa_hdr->fill_len);
 		}
 
 		getbintime(&ooa_hdr->cur_bt);
 
 		if (cur_fill_num > ooa_hdr->alloc_num) {
 			ooa_hdr->dropped_num = cur_fill_num -ooa_hdr->alloc_num;
 			ooa_hdr->status = CTL_OOA_NEED_MORE_SPACE;
 		} else {
 			ooa_hdr->dropped_num = 0;
 			ooa_hdr->status = CTL_OOA_OK;
 		}
 
 		free(entries, M_CTL);
 		break;
 	}
 	case CTL_CHECK_OOA: {
 		union ctl_io *io;
 		struct ctl_lun *lun;
 		struct ctl_ooa_info *ooa_info;
 
 
 		ooa_info = (struct ctl_ooa_info *)addr;
 
 		if (ooa_info->lun_id >= CTL_MAX_LUNS) {
 			ooa_info->status = CTL_OOA_INVALID_LUN;
 			break;
 		}
 		mtx_lock(&softc->ctl_lock);
 		lun = softc->ctl_luns[ooa_info->lun_id];
 		if (lun == NULL) {
 			mtx_unlock(&softc->ctl_lock);
 			ooa_info->status = CTL_OOA_INVALID_LUN;
 			break;
 		}
 		mtx_lock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 		ooa_info->num_entries = 0;
 		for (io = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue);
 		     io != NULL; io = (union ctl_io *)TAILQ_NEXT(
 		     &io->io_hdr, ooa_links)) {
 			ooa_info->num_entries++;
 		}
 		mtx_unlock(&lun->lun_lock);
 
 		ooa_info->status = CTL_OOA_SUCCESS;
 
 		break;
 	}
 	case CTL_HARD_START:
 	case CTL_HARD_STOP: {
 		struct ctl_fe_ioctl_startstop_info ss_info;
 		struct cfi_metatask *metatask;
 		struct mtx hs_mtx;
 
 		mtx_init(&hs_mtx, "HS Mutex", NULL, MTX_DEF);
 
 		cv_init(&ss_info.sem, "hard start/stop cv" );
 
 		metatask = cfi_alloc_metatask(/*can_wait*/ 1);
 		if (metatask == NULL) {
 			retval = ENOMEM;
 			mtx_destroy(&hs_mtx);
 			break;
 		}
 
 		if (cmd == CTL_HARD_START)
 			metatask->tasktype = CFI_TASK_STARTUP;
 		else
 			metatask->tasktype = CFI_TASK_SHUTDOWN;
 
 		metatask->callback = ctl_ioctl_hard_startstop_callback;
 		metatask->callback_arg = &ss_info;
 
 		cfi_action(metatask);
 
 		/* Wait for the callback */
 		mtx_lock(&hs_mtx);
 		cv_wait_sig(&ss_info.sem, &hs_mtx);
 		mtx_unlock(&hs_mtx);
 
 		/*
 		 * All information has been copied from the metatask by the
 		 * time cv_broadcast() is called, so we free the metatask here.
 		 */
 		cfi_free_metatask(metatask);
 
 		memcpy((void *)addr, &ss_info.hs_info, sizeof(ss_info.hs_info));
 
 		mtx_destroy(&hs_mtx);
 		break;
 	}
 	case CTL_BBRREAD: {
 		struct ctl_bbrread_info *bbr_info;
 		struct ctl_fe_ioctl_bbrread_info fe_bbr_info;
 		struct mtx bbr_mtx;
 		struct cfi_metatask *metatask;
 
 		bbr_info = (struct ctl_bbrread_info *)addr;
 
 		bzero(&fe_bbr_info, sizeof(fe_bbr_info));
 
 		bzero(&bbr_mtx, sizeof(bbr_mtx));
 		mtx_init(&bbr_mtx, "BBR Mutex", NULL, MTX_DEF);
 
 		fe_bbr_info.bbr_info = bbr_info;
 		fe_bbr_info.lock = &bbr_mtx;
 
 		cv_init(&fe_bbr_info.sem, "BBR read cv");
 		metatask = cfi_alloc_metatask(/*can_wait*/ 1);
 
 		if (metatask == NULL) {
 			mtx_destroy(&bbr_mtx);
 			cv_destroy(&fe_bbr_info.sem);
 			retval = ENOMEM;
 			break;
 		}
 		metatask->tasktype = CFI_TASK_BBRREAD;
 		metatask->callback = ctl_ioctl_bbrread_callback;
 		metatask->callback_arg = &fe_bbr_info;
 		metatask->taskinfo.bbrread.lun_num = bbr_info->lun_num;
 		metatask->taskinfo.bbrread.lba = bbr_info->lba;
 		metatask->taskinfo.bbrread.len = bbr_info->len;
 
 		cfi_action(metatask);
 
 		mtx_lock(&bbr_mtx);
 		while (fe_bbr_info.wakeup_done == 0)
 			cv_wait_sig(&fe_bbr_info.sem, &bbr_mtx);
 		mtx_unlock(&bbr_mtx);
 
 		bbr_info->status = metatask->status;
 		bbr_info->bbr_status = metatask->taskinfo.bbrread.status;
 		bbr_info->scsi_status = metatask->taskinfo.bbrread.scsi_status;
 		memcpy(&bbr_info->sense_data,
 		       &metatask->taskinfo.bbrread.sense_data,
 		       MIN(sizeof(bbr_info->sense_data),
 			   sizeof(metatask->taskinfo.bbrread.sense_data)));
 
 		cfi_free_metatask(metatask);
 
 		mtx_destroy(&bbr_mtx);
 		cv_destroy(&fe_bbr_info.sem);
 
 		break;
 	}
 	case CTL_DELAY_IO: {
 		struct ctl_io_delay_info *delay_info;
 #ifdef CTL_IO_DELAY
 		struct ctl_lun *lun;
 #endif /* CTL_IO_DELAY */
 
 		delay_info = (struct ctl_io_delay_info *)addr;
 
 #ifdef CTL_IO_DELAY
 		mtx_lock(&softc->ctl_lock);
 
 		if ((delay_info->lun_id >= CTL_MAX_LUNS)
 		 || (softc->ctl_luns[delay_info->lun_id] == NULL)) {
 			delay_info->status = CTL_DELAY_STATUS_INVALID_LUN;
 		} else {
 			lun = softc->ctl_luns[delay_info->lun_id];
 			mtx_lock(&lun->lun_lock);
 
 			delay_info->status = CTL_DELAY_STATUS_OK;
 
 			switch (delay_info->delay_type) {
 			case CTL_DELAY_TYPE_CONT:
 				break;
 			case CTL_DELAY_TYPE_ONESHOT:
 				break;
 			default:
 				delay_info->status =
 					CTL_DELAY_STATUS_INVALID_TYPE;
 				break;
 			}
 
 			switch (delay_info->delay_loc) {
 			case CTL_DELAY_LOC_DATAMOVE:
 				lun->delay_info.datamove_type =
 					delay_info->delay_type;
 				lun->delay_info.datamove_delay =
 					delay_info->delay_secs;
 				break;
 			case CTL_DELAY_LOC_DONE:
 				lun->delay_info.done_type =
 					delay_info->delay_type;
 				lun->delay_info.done_delay =
 					delay_info->delay_secs;
 				break;
 			default:
 				delay_info->status =
 					CTL_DELAY_STATUS_INVALID_LOC;
 				break;
 			}
 			mtx_unlock(&lun->lun_lock);
 		}
 
 		mtx_unlock(&softc->ctl_lock);
 #else
 		delay_info->status = CTL_DELAY_STATUS_NOT_IMPLEMENTED;
 #endif /* CTL_IO_DELAY */
 		break;
 	}
 	case CTL_REALSYNC_SET: {
 		int *syncstate;
 
 		syncstate = (int *)addr;
 
 		mtx_lock(&softc->ctl_lock);
 		switch (*syncstate) {
 		case 0:
 			softc->flags &= ~CTL_FLAG_REAL_SYNC;
 			break;
 		case 1:
 			softc->flags |= CTL_FLAG_REAL_SYNC;
 			break;
 		default:
 			retval = EINVAL;
 			break;
 		}
 		mtx_unlock(&softc->ctl_lock);
 		break;
 	}
 	case CTL_REALSYNC_GET: {
 		int *syncstate;
 
 		syncstate = (int*)addr;
 
 		mtx_lock(&softc->ctl_lock);
 		if (softc->flags & CTL_FLAG_REAL_SYNC)
 			*syncstate = 1;
 		else
 			*syncstate = 0;
 		mtx_unlock(&softc->ctl_lock);
 
 		break;
 	}
 	case CTL_SETSYNC:
 	case CTL_GETSYNC: {
 		struct ctl_sync_info *sync_info;
 		struct ctl_lun *lun;
 
 		sync_info = (struct ctl_sync_info *)addr;
 
 		mtx_lock(&softc->ctl_lock);
 		lun = softc->ctl_luns[sync_info->lun_id];
 		if (lun == NULL) {
 			mtx_unlock(&softc->ctl_lock);
 			sync_info->status = CTL_GS_SYNC_NO_LUN;
 		}
 		/*
 		 * Get or set the sync interval.  We're not bounds checking
 		 * in the set case, hopefully the user won't do something
 		 * silly.
 		 */
 		mtx_lock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 		if (cmd == CTL_GETSYNC)
 			sync_info->sync_interval = lun->sync_interval;
 		else
 			lun->sync_interval = sync_info->sync_interval;
 		mtx_unlock(&lun->lun_lock);
 
 		sync_info->status = CTL_GS_SYNC_OK;
 
 		break;
 	}
 	case CTL_GETSTATS: {
 		struct ctl_stats *stats;
 		struct ctl_lun *lun;
 		int i;
 
 		stats = (struct ctl_stats *)addr;
 
 		if ((sizeof(struct ctl_lun_io_stats) * softc->num_luns) >
 		     stats->alloc_len) {
 			stats->status = CTL_SS_NEED_MORE_SPACE;
 			stats->num_luns = softc->num_luns;
 			break;
 		}
 		/*
 		 * XXX KDM no locking here.  If the LUN list changes,
 		 * things can blow up.
 		 */
 		for (i = 0, lun = STAILQ_FIRST(&softc->lun_list); lun != NULL;
 		     i++, lun = STAILQ_NEXT(lun, links)) {
 			retval = copyout(&lun->stats, &stats->lun_stats[i],
 					 sizeof(lun->stats));
 			if (retval != 0)
 				break;
 		}
 		stats->num_luns = softc->num_luns;
 		stats->fill_len = sizeof(struct ctl_lun_io_stats) *
 				 softc->num_luns;
 		stats->status = CTL_SS_OK;
 #ifdef CTL_TIME_IO
 		stats->flags = CTL_STATS_FLAG_TIME_VALID;
 #else
 		stats->flags = CTL_STATS_FLAG_NONE;
 #endif
 		getnanouptime(&stats->timestamp);
 		break;
 	}
 	case CTL_ERROR_INJECT: {
 		struct ctl_error_desc *err_desc, *new_err_desc;
 		struct ctl_lun *lun;
 
 		err_desc = (struct ctl_error_desc *)addr;
 
 		new_err_desc = malloc(sizeof(*new_err_desc), M_CTL,
 				      M_WAITOK | M_ZERO);
 		bcopy(err_desc, new_err_desc, sizeof(*new_err_desc));
 
 		mtx_lock(&softc->ctl_lock);
 		lun = softc->ctl_luns[err_desc->lun_id];
 		if (lun == NULL) {
 			mtx_unlock(&softc->ctl_lock);
 			free(new_err_desc, M_CTL);
 			printf("%s: CTL_ERROR_INJECT: invalid LUN %ju\n",
 			       __func__, (uintmax_t)err_desc->lun_id);
 			retval = EINVAL;
 			break;
 		}
 		mtx_lock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 
 		/*
 		 * We could do some checking here to verify the validity
 		 * of the request, but given the complexity of error
 		 * injection requests, the checking logic would be fairly
 		 * complex.
 		 *
 		 * For now, if the request is invalid, it just won't get
 		 * executed and might get deleted.
 		 */
 		STAILQ_INSERT_TAIL(&lun->error_list, new_err_desc, links);
 
 		/*
 		 * XXX KDM check to make sure the serial number is unique,
 		 * in case we somehow manage to wrap.  That shouldn't
 		 * happen for a very long time, but it's the right thing to
 		 * do.
 		 */
 		new_err_desc->serial = lun->error_serial;
 		err_desc->serial = lun->error_serial;
 		lun->error_serial++;
 
 		mtx_unlock(&lun->lun_lock);
 		break;
 	}
 	case CTL_ERROR_INJECT_DELETE: {
 		struct ctl_error_desc *delete_desc, *desc, *desc2;
 		struct ctl_lun *lun;
 		int delete_done;
 
 		delete_desc = (struct ctl_error_desc *)addr;
 		delete_done = 0;
 
 		mtx_lock(&softc->ctl_lock);
 		lun = softc->ctl_luns[delete_desc->lun_id];
 		if (lun == NULL) {
 			mtx_unlock(&softc->ctl_lock);
 			printf("%s: CTL_ERROR_INJECT_DELETE: invalid LUN %ju\n",
 			       __func__, (uintmax_t)delete_desc->lun_id);
 			retval = EINVAL;
 			break;
 		}
 		mtx_lock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 		STAILQ_FOREACH_SAFE(desc, &lun->error_list, links, desc2) {
 			if (desc->serial != delete_desc->serial)
 				continue;
 
 			STAILQ_REMOVE(&lun->error_list, desc, ctl_error_desc,
 				      links);
 			free(desc, M_CTL);
 			delete_done = 1;
 		}
 		mtx_unlock(&lun->lun_lock);
 		if (delete_done == 0) {
 			printf("%s: CTL_ERROR_INJECT_DELETE: can't find "
 			       "error serial %ju on LUN %u\n", __func__, 
 			       delete_desc->serial, delete_desc->lun_id);
 			retval = EINVAL;
 			break;
 		}
 		break;
 	}
 	case CTL_DUMP_STRUCTS: {
 		int i, j, k;
 		struct ctl_port *port;
 		struct ctl_frontend *fe;
 
 		mtx_lock(&softc->ctl_lock);
 		printf("CTL Persistent Reservation information start:\n");
 		for (i = 0; i < CTL_MAX_LUNS; i++) {
 			struct ctl_lun *lun;
 
 			lun = softc->ctl_luns[i];
 
 			if ((lun == NULL)
 			 || ((lun->flags & CTL_LUN_DISABLED) != 0))
 				continue;
 
 			for (j = 0; j < (CTL_MAX_PORTS * 2); j++) {
 				if (lun->pr_keys[j] == NULL)
 					continue;
 				for (k = 0; k < CTL_MAX_INIT_PER_PORT; k++){
 					if (lun->pr_keys[j][k] == 0)
 						continue;
 					printf("  LUN %d port %d iid %d key "
 					       "%#jx\n", i, j, k,
 					       (uintmax_t)lun->pr_keys[j][k]);
 				}
 			}
 		}
 		printf("CTL Persistent Reservation information end\n");
 		printf("CTL Ports:\n");
 		STAILQ_FOREACH(port, &softc->port_list, links) {
 			printf("  Port %d '%s' Frontend '%s' Type %u pp %d vp %d WWNN "
 			       "%#jx WWPN %#jx\n", port->targ_port, port->port_name,
 			       port->frontend->name, port->port_type,
 			       port->physical_port, port->virtual_port,
 			       (uintmax_t)port->wwnn, (uintmax_t)port->wwpn);
 			for (j = 0; j < CTL_MAX_INIT_PER_PORT; j++) {
 				if (port->wwpn_iid[j].in_use == 0 &&
 				    port->wwpn_iid[j].wwpn == 0 &&
 				    port->wwpn_iid[j].name == NULL)
 					continue;
 
 				printf("    iid %u use %d WWPN %#jx '%s'\n",
 				    j, port->wwpn_iid[j].in_use,
 				    (uintmax_t)port->wwpn_iid[j].wwpn,
 				    port->wwpn_iid[j].name);
 			}
 		}
 		printf("CTL Port information end\n");
 		mtx_unlock(&softc->ctl_lock);
 		/*
 		 * XXX KDM calling this without a lock.  We'd likely want
 		 * to drop the lock before calling the frontend's dump
 		 * routine anyway.
 		 */
 		printf("CTL Frontends:\n");
 		STAILQ_FOREACH(fe, &softc->fe_list, links) {
 			printf("  Frontend '%s'\n", fe->name);
 			if (fe->fe_dump != NULL)
 				fe->fe_dump();
 		}
 		printf("CTL Frontend information end\n");
 		break;
 	}
 	case CTL_LUN_REQ: {
 		struct ctl_lun_req *lun_req;
 		struct ctl_backend_driver *backend;
 
 		lun_req = (struct ctl_lun_req *)addr;
 
 		backend = ctl_backend_find(lun_req->backend);
 		if (backend == NULL) {
 			lun_req->status = CTL_LUN_ERROR;
 			snprintf(lun_req->error_str,
 				 sizeof(lun_req->error_str),
 				 "Backend \"%s\" not found.",
 				 lun_req->backend);
 			break;
 		}
 		if (lun_req->num_be_args > 0) {
 			lun_req->kern_be_args = ctl_copyin_args(
 				lun_req->num_be_args,
 				lun_req->be_args,
 				lun_req->error_str,
 				sizeof(lun_req->error_str));
 			if (lun_req->kern_be_args == NULL) {
 				lun_req->status = CTL_LUN_ERROR;
 				break;
 			}
 		}
 
 		retval = backend->ioctl(dev, cmd, addr, flag, td);
 
 		if (lun_req->num_be_args > 0) {
 			ctl_copyout_args(lun_req->num_be_args,
 				      lun_req->kern_be_args);
 			ctl_free_args(lun_req->num_be_args,
 				      lun_req->kern_be_args);
 		}
 		break;
 	}
 	case CTL_LUN_LIST: {
 		struct sbuf *sb;
 		struct ctl_lun *lun;
 		struct ctl_lun_list *list;
 		struct ctl_option *opt;
 
 		list = (struct ctl_lun_list *)addr;
 
 		/*
 		 * Allocate a fixed length sbuf here, based on the length
 		 * of the user's buffer.  We could allocate an auto-extending
 		 * buffer, and then tell the user how much larger our
 		 * amount of data is than his buffer, but that presents
 		 * some problems:
 		 *
 		 * 1.  The sbuf(9) routines use a blocking malloc, and so
 		 *     we can't hold a lock while calling them with an
 		 *     auto-extending buffer.
  		 *
 		 * 2.  There is not currently a LUN reference counting
 		 *     mechanism, outside of outstanding transactions on
 		 *     the LUN's OOA queue.  So a LUN could go away on us
 		 *     while we're getting the LUN number, backend-specific
 		 *     information, etc.  Thus, given the way things
 		 *     currently work, we need to hold the CTL lock while
 		 *     grabbing LUN information.
 		 *
 		 * So, from the user's standpoint, the best thing to do is
 		 * allocate what he thinks is a reasonable buffer length,
 		 * and then if he gets a CTL_LUN_LIST_NEED_MORE_SPACE error,
 		 * double the buffer length and try again.  (And repeat
 		 * that until he succeeds.)
 		 */
 		sb = sbuf_new(NULL, NULL, list->alloc_len, SBUF_FIXEDLEN);
 		if (sb == NULL) {
 			list->status = CTL_LUN_LIST_ERROR;
 			snprintf(list->error_str, sizeof(list->error_str),
 				 "Unable to allocate %d bytes for LUN list",
 				 list->alloc_len);
 			break;
 		}
 
 		sbuf_printf(sb, "<ctllunlist>\n");
 
 		mtx_lock(&softc->ctl_lock);
 		STAILQ_FOREACH(lun, &softc->lun_list, links) {
 			mtx_lock(&lun->lun_lock);
 			retval = sbuf_printf(sb, "<lun id=\"%ju\">\n",
 					     (uintmax_t)lun->lun);
 
 			/*
 			 * Bail out as soon as we see that we've overfilled
 			 * the buffer.
 			 */
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<backend_type>%s"
 					     "</backend_type>\n",
 					     (lun->backend == NULL) ?  "none" :
 					     lun->backend->name);
 
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<lun_type>%d</lun_type>\n",
 					     lun->be_lun->lun_type);
 
 			if (retval != 0)
 				break;
 
 			if (lun->backend == NULL) {
 				retval = sbuf_printf(sb, "</lun>\n");
 				if (retval != 0)
 					break;
 				continue;
 			}
 
 			retval = sbuf_printf(sb, "\t<size>%ju</size>\n",
 					     (lun->be_lun->maxlba > 0) ?
 					     lun->be_lun->maxlba + 1 : 0);
 
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<blocksize>%u</blocksize>\n",
 					     lun->be_lun->blocksize);
 
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<serial_number>");
 
 			if (retval != 0)
 				break;
 
 			retval = ctl_sbuf_printf_esc(sb,
 			    lun->be_lun->serial_num,
 			    sizeof(lun->be_lun->serial_num));
 
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "</serial_number>\n");
 		
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<device_id>");
 
 			if (retval != 0)
 				break;
 
 			retval = ctl_sbuf_printf_esc(sb,
 			    lun->be_lun->device_id,
 			    sizeof(lun->be_lun->device_id));
 
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "</device_id>\n");
 
 			if (retval != 0)
 				break;
 
 			if (lun->backend->lun_info != NULL) {
 				retval = lun->backend->lun_info(lun->be_lun->be_lun, sb);
 				if (retval != 0)
 					break;
 			}
 			STAILQ_FOREACH(opt, &lun->be_lun->options, links) {
 				retval = sbuf_printf(sb, "\t<%s>%s</%s>\n",
 				    opt->name, opt->value, opt->name);
 				if (retval != 0)
 					break;
 			}
 
 			retval = sbuf_printf(sb, "</lun>\n");
 
 			if (retval != 0)
 				break;
 			mtx_unlock(&lun->lun_lock);
 		}
 		if (lun != NULL)
 			mtx_unlock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 
 		if ((retval != 0)
 		 || ((retval = sbuf_printf(sb, "</ctllunlist>\n")) != 0)) {
 			retval = 0;
 			sbuf_delete(sb);
 			list->status = CTL_LUN_LIST_NEED_MORE_SPACE;
 			snprintf(list->error_str, sizeof(list->error_str),
 				 "Out of space, %d bytes is too small",
 				 list->alloc_len);
 			break;
 		}
 
 		sbuf_finish(sb);
 
 		retval = copyout(sbuf_data(sb), list->lun_xml,
 				 sbuf_len(sb) + 1);
 
 		list->fill_len = sbuf_len(sb) + 1;
 		list->status = CTL_LUN_LIST_OK;
 		sbuf_delete(sb);
 		break;
 	}
 	case CTL_ISCSI: {
 		struct ctl_iscsi *ci;
 		struct ctl_frontend *fe;
 
 		ci = (struct ctl_iscsi *)addr;
 
 		fe = ctl_frontend_find("iscsi");
 		if (fe == NULL) {
 			ci->status = CTL_ISCSI_ERROR;
 			snprintf(ci->error_str, sizeof(ci->error_str),
 			    "Frontend \"iscsi\" not found.");
 			break;
 		}
 
 		retval = fe->ioctl(dev, cmd, addr, flag, td);
 		break;
 	}
 	case CTL_PORT_REQ: {
 		struct ctl_req *req;
 		struct ctl_frontend *fe;
 
 		req = (struct ctl_req *)addr;
 
 		fe = ctl_frontend_find(req->driver);
 		if (fe == NULL) {
 			req->status = CTL_LUN_ERROR;
 			snprintf(req->error_str, sizeof(req->error_str),
 			    "Frontend \"%s\" not found.", req->driver);
 			break;
 		}
 		if (req->num_args > 0) {
 			req->kern_args = ctl_copyin_args(req->num_args,
 			    req->args, req->error_str, sizeof(req->error_str));
 			if (req->kern_args == NULL) {
 				req->status = CTL_LUN_ERROR;
 				break;
 			}
 		}
 
 		retval = fe->ioctl(dev, cmd, addr, flag, td);
 
 		if (req->num_args > 0) {
 			ctl_copyout_args(req->num_args, req->kern_args);
 			ctl_free_args(req->num_args, req->kern_args);
 		}
 		break;
 	}
 	case CTL_PORT_LIST: {
 		struct sbuf *sb;
 		struct ctl_port *port;
 		struct ctl_lun_list *list;
 		struct ctl_option *opt;
 		int j;
 
 		list = (struct ctl_lun_list *)addr;
 
 		sb = sbuf_new(NULL, NULL, list->alloc_len, SBUF_FIXEDLEN);
 		if (sb == NULL) {
 			list->status = CTL_LUN_LIST_ERROR;
 			snprintf(list->error_str, sizeof(list->error_str),
 				 "Unable to allocate %d bytes for LUN list",
 				 list->alloc_len);
 			break;
 		}
 
 		sbuf_printf(sb, "<ctlportlist>\n");
 
 		mtx_lock(&softc->ctl_lock);
 		STAILQ_FOREACH(port, &softc->port_list, links) {
 			retval = sbuf_printf(sb, "<targ_port id=\"%ju\">\n",
 					     (uintmax_t)port->targ_port);
 
 			/*
 			 * Bail out as soon as we see that we've overfilled
 			 * the buffer.
 			 */
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<frontend_type>%s"
 			    "</frontend_type>\n", port->frontend->name);
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<port_type>%d</port_type>\n",
 					     port->port_type);
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<online>%s</online>\n",
 			    (port->status & CTL_PORT_STATUS_ONLINE) ? "YES" : "NO");
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<port_name>%s</port_name>\n",
 			    port->port_name);
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<physical_port>%d</physical_port>\n",
 			    port->physical_port);
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "\t<virtual_port>%d</virtual_port>\n",
 			    port->virtual_port);
 			if (retval != 0)
 				break;
 
 			if (port->target_devid != NULL) {
 				sbuf_printf(sb, "\t<target>");
 				ctl_id_sbuf(port->target_devid, sb);
 				sbuf_printf(sb, "</target>\n");
 			}
 
 			if (port->port_devid != NULL) {
 				sbuf_printf(sb, "\t<port>");
 				ctl_id_sbuf(port->port_devid, sb);
 				sbuf_printf(sb, "</port>\n");
 			}
 
 			if (port->port_info != NULL) {
 				retval = port->port_info(port->onoff_arg, sb);
 				if (retval != 0)
 					break;
 			}
 			STAILQ_FOREACH(opt, &port->options, links) {
 				retval = sbuf_printf(sb, "\t<%s>%s</%s>\n",
 				    opt->name, opt->value, opt->name);
 				if (retval != 0)
 					break;
 			}
 
 			for (j = 0; j < CTL_MAX_INIT_PER_PORT; j++) {
 				if (port->wwpn_iid[j].in_use == 0 ||
 				    (port->wwpn_iid[j].wwpn == 0 &&
 				     port->wwpn_iid[j].name == NULL))
 					continue;
 
 				if (port->wwpn_iid[j].name != NULL)
 					retval = sbuf_printf(sb,
 					    "\t<initiator id=\"%u\">%s</initiator>\n",
 					    j, port->wwpn_iid[j].name);
 				else
 					retval = sbuf_printf(sb,
 					    "\t<initiator id=\"%u\">naa.%08jx</initiator>\n",
 					    j, port->wwpn_iid[j].wwpn);
 				if (retval != 0)
 					break;
 			}
 			if (retval != 0)
 				break;
 
 			retval = sbuf_printf(sb, "</targ_port>\n");
 			if (retval != 0)
 				break;
 		}
 		mtx_unlock(&softc->ctl_lock);
 
 		if ((retval != 0)
 		 || ((retval = sbuf_printf(sb, "</ctlportlist>\n")) != 0)) {
 			retval = 0;
 			sbuf_delete(sb);
 			list->status = CTL_LUN_LIST_NEED_MORE_SPACE;
 			snprintf(list->error_str, sizeof(list->error_str),
 				 "Out of space, %d bytes is too small",
 				 list->alloc_len);
 			break;
 		}
 
 		sbuf_finish(sb);
 
 		retval = copyout(sbuf_data(sb), list->lun_xml,
 				 sbuf_len(sb) + 1);
 
 		list->fill_len = sbuf_len(sb) + 1;
 		list->status = CTL_LUN_LIST_OK;
 		sbuf_delete(sb);
 		break;
 	}
 	default: {
 		/* XXX KDM should we fix this? */
 #if 0
 		struct ctl_backend_driver *backend;
 		unsigned int type;
 		int found;
 
 		found = 0;
 
 		/*
 		 * We encode the backend type as the ioctl type for backend
 		 * ioctls.  So parse it out here, and then search for a
 		 * backend of this type.
 		 */
 		type = _IOC_TYPE(cmd);
 
 		STAILQ_FOREACH(backend, &softc->be_list, links) {
 			if (backend->type == type) {
 				found = 1;
 				break;
 			}
 		}
 		if (found == 0) {
 			printf("ctl: unknown ioctl command %#lx or backend "
 			       "%d\n", cmd, type);
 			retval = EINVAL;
 			break;
 		}
 		retval = backend->ioctl(dev, cmd, addr, flag, td);
 #endif
 		retval = ENOTTY;
 		break;
 	}
 	}
 	return (retval);
 }
 
 uint32_t
 ctl_get_initindex(struct ctl_nexus *nexus)
 {
 	if (nexus->targ_port < CTL_MAX_PORTS)
 		return (nexus->initid.id +
 			(nexus->targ_port * CTL_MAX_INIT_PER_PORT));
 	else
 		return (nexus->initid.id +
 		       ((nexus->targ_port - CTL_MAX_PORTS) *
 			CTL_MAX_INIT_PER_PORT));
 }
 
 uint32_t
 ctl_get_resindex(struct ctl_nexus *nexus)
 {
 	return (nexus->initid.id + (nexus->targ_port * CTL_MAX_INIT_PER_PORT));
 }
 
 uint32_t
 ctl_port_idx(int port_num)
 {
 	if (port_num < CTL_MAX_PORTS)
 		return(port_num);
 	else
 		return(port_num - CTL_MAX_PORTS);
 }
 
 static uint32_t
 ctl_map_lun(struct ctl_softc *softc, int port_num, uint32_t lun_id)
 {
 	struct ctl_port *port;
 
 	port = softc->ctl_ports[ctl_port_idx(port_num)];
 	if (port == NULL)
 		return (UINT32_MAX);
 	if (port->lun_map == NULL)
 		return (lun_id);
 	return (port->lun_map(port->targ_lun_arg, lun_id));
 }
 
 static uint32_t
 ctl_map_lun_back(struct ctl_softc *softc, int port_num, uint32_t lun_id)
 {
 	struct ctl_port *port;
 	uint32_t i;
 
 	port = softc->ctl_ports[ctl_port_idx(port_num)];
 	if (port->lun_map == NULL)
 		return (lun_id);
 	for (i = 0; i < CTL_MAX_LUNS; i++) {
 		if (port->lun_map(port->targ_lun_arg, i) == lun_id)
 			return (i);
 	}
 	return (UINT32_MAX);
 }
 
 /*
  * Note:  This only works for bitmask sizes that are at least 32 bits, and
  * that are a power of 2.
  */
 int
 ctl_ffz(uint32_t *mask, uint32_t size)
 {
 	uint32_t num_chunks, num_pieces;
 	int i, j;
 
 	num_chunks = (size >> 5);
 	if (num_chunks == 0)
 		num_chunks++;
 	num_pieces = MIN((sizeof(uint32_t) * 8), size);
 
 	for (i = 0; i < num_chunks; i++) {
 		for (j = 0; j < num_pieces; j++) {
 			if ((mask[i] & (1 << j)) == 0)
 				return ((i << 5) + j);
 		}
 	}
 
 	return (-1);
 }
 
 int
 ctl_set_mask(uint32_t *mask, uint32_t bit)
 {
 	uint32_t chunk, piece;
 
 	chunk = bit >> 5;
 	piece = bit % (sizeof(uint32_t) * 8);
 
 	if ((mask[chunk] & (1 << piece)) != 0)
 		return (-1);
 	else
 		mask[chunk] |= (1 << piece);
 
 	return (0);
 }
 
 int
 ctl_clear_mask(uint32_t *mask, uint32_t bit)
 {
 	uint32_t chunk, piece;
 
 	chunk = bit >> 5;
 	piece = bit % (sizeof(uint32_t) * 8);
 
 	if ((mask[chunk] & (1 << piece)) == 0)
 		return (-1);
 	else
 		mask[chunk] &= ~(1 << piece);
 
 	return (0);
 }
 
 int
 ctl_is_set(uint32_t *mask, uint32_t bit)
 {
 	uint32_t chunk, piece;
 
 	chunk = bit >> 5;
 	piece = bit % (sizeof(uint32_t) * 8);
 
 	if ((mask[chunk] & (1 << piece)) == 0)
 		return (0);
 	else
 		return (1);
 }
 
 static uint64_t
 ctl_get_prkey(struct ctl_lun *lun, uint32_t residx)
 {
 	uint64_t *t;
 
 	t = lun->pr_keys[residx/CTL_MAX_INIT_PER_PORT];
 	if (t == NULL)
 		return (0);
 	return (t[residx % CTL_MAX_INIT_PER_PORT]);
 }
 
 static void
 ctl_clr_prkey(struct ctl_lun *lun, uint32_t residx)
 {
 	uint64_t *t;
 
 	t = lun->pr_keys[residx/CTL_MAX_INIT_PER_PORT];
 	if (t == NULL)
 		return;
 	t[residx % CTL_MAX_INIT_PER_PORT] = 0;
 }
 
 static void
 ctl_alloc_prkey(struct ctl_lun *lun, uint32_t residx)
 {
 	uint64_t *p;
 	u_int i;
 
 	i = residx/CTL_MAX_INIT_PER_PORT;
 	if (lun->pr_keys[i] != NULL)
 		return;
 	mtx_unlock(&lun->lun_lock);
 	p = malloc(sizeof(uint64_t) * CTL_MAX_INIT_PER_PORT, M_CTL,
 	    M_WAITOK | M_ZERO);
 	mtx_lock(&lun->lun_lock);
 	if (lun->pr_keys[i] == NULL)
 		lun->pr_keys[i] = p;
 	else
 		free(p, M_CTL);
 }
 
 static void
 ctl_set_prkey(struct ctl_lun *lun, uint32_t residx, uint64_t key)
 {
 	uint64_t *t;
 
 	t = lun->pr_keys[residx/CTL_MAX_INIT_PER_PORT];
 	KASSERT(t != NULL, ("prkey %d is not allocated", residx));
 	t[residx % CTL_MAX_INIT_PER_PORT] = key;
 }
 
 /*
  * ctl_softc, pool_name, total_ctl_io are passed in.
  * npool is passed out.
  */
 int
 ctl_pool_create(struct ctl_softc *ctl_softc, const char *pool_name,
 		uint32_t total_ctl_io, void **npool)
 {
 #ifdef IO_POOLS
 	struct ctl_io_pool *pool;
 
 	pool = (struct ctl_io_pool *)malloc(sizeof(*pool), M_CTL,
 					    M_NOWAIT | M_ZERO);
 	if (pool == NULL)
 		return (ENOMEM);
 
 	snprintf(pool->name, sizeof(pool->name), "CTL IO %s", pool_name);
 	pool->ctl_softc = ctl_softc;
 	pool->zone = uma_zsecond_create(pool->name, NULL,
 	    NULL, NULL, NULL, ctl_softc->io_zone);
 	/* uma_prealloc(pool->zone, total_ctl_io); */
 
 	*npool = pool;
 #else
 	*npool = ctl_softc->io_zone;
 #endif
 	return (0);
 }
 
 void
 ctl_pool_free(struct ctl_io_pool *pool)
 {
 
 	if (pool == NULL)
 		return;
 
 #ifdef IO_POOLS
 	uma_zdestroy(pool->zone);
 	free(pool, M_CTL);
 #endif
 }
 
 union ctl_io *
 ctl_alloc_io(void *pool_ref)
 {
 	union ctl_io *io;
 #ifdef IO_POOLS
 	struct ctl_io_pool *pool = (struct ctl_io_pool *)pool_ref;
 
 	io = uma_zalloc(pool->zone, M_WAITOK);
 #else
 	io = uma_zalloc((uma_zone_t)pool_ref, M_WAITOK);
 #endif
 	if (io != NULL)
 		io->io_hdr.pool = pool_ref;
 	return (io);
 }
 
 union ctl_io *
 ctl_alloc_io_nowait(void *pool_ref)
 {
 	union ctl_io *io;
 #ifdef IO_POOLS
 	struct ctl_io_pool *pool = (struct ctl_io_pool *)pool_ref;
 
 	io = uma_zalloc(pool->zone, M_NOWAIT);
 #else
 	io = uma_zalloc((uma_zone_t)pool_ref, M_NOWAIT);
 #endif
 	if (io != NULL)
 		io->io_hdr.pool = pool_ref;
 	return (io);
 }
 
 void
 ctl_free_io(union ctl_io *io)
 {
 #ifdef IO_POOLS
 	struct ctl_io_pool *pool;
 #endif
 
 	if (io == NULL)
 		return;
 
 #ifdef IO_POOLS
 	pool = (struct ctl_io_pool *)io->io_hdr.pool;
 	uma_zfree(pool->zone, io);
 #else
 	uma_zfree((uma_zone_t)io->io_hdr.pool, io);
 #endif
 }
 
 void
 ctl_zero_io(union ctl_io *io)
 {
 	void *pool_ref;
 
 	if (io == NULL)
 		return;
 
 	/*
 	 * May need to preserve linked list pointers at some point too.
 	 */
 	pool_ref = io->io_hdr.pool;
 	memset(io, 0, sizeof(*io));
 	io->io_hdr.pool = pool_ref;
 }
 
 /*
  * This routine is currently used for internal copies of ctl_ios that need
  * to persist for some reason after we've already returned status to the
  * FETD.  (Thus the flag set.)
  *
  * XXX XXX
  * Note that this makes a blind copy of all fields in the ctl_io, except
  * for the pool reference.  This includes any memory that has been
  * allocated!  That memory will no longer be valid after done has been
  * called, so this would be VERY DANGEROUS for command that actually does
  * any reads or writes.  Right now (11/7/2005), this is only used for immediate
  * start and stop commands, which don't transfer any data, so this is not a
  * problem.  If it is used for anything else, the caller would also need to
  * allocate data buffer space and this routine would need to be modified to
  * copy the data buffer(s) as well.
  */
 void
 ctl_copy_io(union ctl_io *src, union ctl_io *dest)
 {
 	void *pool_ref;
 
 	if ((src == NULL)
 	 || (dest == NULL))
 		return;
 
 	/*
 	 * May need to preserve linked list pointers at some point too.
 	 */
 	pool_ref = dest->io_hdr.pool;
 
 	memcpy(dest, src, MIN(sizeof(*src), sizeof(*dest)));
 
 	dest->io_hdr.pool = pool_ref;
 	/*
 	 * We need to know that this is an internal copy, and doesn't need
 	 * to get passed back to the FETD that allocated it.
 	 */
 	dest->io_hdr.flags |= CTL_FLAG_INT_COPY;
 }
 
 int
 ctl_expand_number(const char *buf, uint64_t *num)
 {
 	char *endptr;
 	uint64_t number;
 	unsigned shift;
 
 	number = strtoq(buf, &endptr, 0);
 
 	switch (tolower((unsigned char)*endptr)) {
 	case 'e':
 		shift = 60;
 		break;
 	case 'p':
 		shift = 50;
 		break;
 	case 't':
 		shift = 40;
 		break;
 	case 'g':
 		shift = 30;
 		break;
 	case 'm':
 		shift = 20;
 		break;
 	case 'k':
 		shift = 10;
 		break;
 	case 'b':
 	case '\0': /* No unit. */
 		*num = number;
 		return (0);
 	default:
 		/* Unrecognized unit. */
 		return (-1);
 	}
 
 	if ((number << shift) >> shift != number) {
 		/* Overflow */
 		return (-1);
 	}
 	*num = number << shift;
 	return (0);
 }
 
 
 /*
  * This routine could be used in the future to load default and/or saved
  * mode page parameters for a particuar lun.
  */
 static int
 ctl_init_page_index(struct ctl_lun *lun)
 {
 	int i;
 	struct ctl_page_index *page_index;
 	const char *value;
 	uint64_t ival;
 
 	memcpy(&lun->mode_pages.index, page_index_template,
 	       sizeof(page_index_template));
 
 	for (i = 0; i < CTL_NUM_MODE_PAGES; i++) {
 
 		page_index = &lun->mode_pages.index[i];
 		/*
 		 * If this is a disk-only mode page, there's no point in
 		 * setting it up.  For some pages, we have to have some
 		 * basic information about the disk in order to calculate the
 		 * mode page data.
 		 */
 		if ((lun->be_lun->lun_type != T_DIRECT)
 		 && (page_index->page_flags & CTL_PAGE_FLAG_DISK_ONLY))
 			continue;
 
 		switch (page_index->page_code & SMPH_PC_MASK) {
 		case SMS_RW_ERROR_RECOVERY_PAGE: {
 			if (page_index->subpage != SMS_SUBPAGE_PAGE_0)
 				panic("subpage is incorrect!");
 			memcpy(&lun->mode_pages.rw_er_page[CTL_PAGE_CURRENT],
 			       &rw_er_page_default,
 			       sizeof(rw_er_page_default));
 			memcpy(&lun->mode_pages.rw_er_page[CTL_PAGE_CHANGEABLE],
 			       &rw_er_page_changeable,
 			       sizeof(rw_er_page_changeable));
 			memcpy(&lun->mode_pages.rw_er_page[CTL_PAGE_DEFAULT],
 			       &rw_er_page_default,
 			       sizeof(rw_er_page_default));
 			memcpy(&lun->mode_pages.rw_er_page[CTL_PAGE_SAVED],
 			       &rw_er_page_default,
 			       sizeof(rw_er_page_default));
 			page_index->page_data =
 				(uint8_t *)lun->mode_pages.rw_er_page;
 			break;
 		}
 		case SMS_FORMAT_DEVICE_PAGE: {
 			struct scsi_format_page *format_page;
 
 			if (page_index->subpage != SMS_SUBPAGE_PAGE_0)
 				panic("subpage is incorrect!");
 
 			/*
 			 * Sectors per track are set above.  Bytes per
 			 * sector need to be set here on a per-LUN basis.
 			 */
 			memcpy(&lun->mode_pages.format_page[CTL_PAGE_CURRENT],
 			       &format_page_default,
 			       sizeof(format_page_default));
 			memcpy(&lun->mode_pages.format_page[
 			       CTL_PAGE_CHANGEABLE], &format_page_changeable,
 			       sizeof(format_page_changeable));
 			memcpy(&lun->mode_pages.format_page[CTL_PAGE_DEFAULT],
 			       &format_page_default,
 			       sizeof(format_page_default));
 			memcpy(&lun->mode_pages.format_page[CTL_PAGE_SAVED],
 			       &format_page_default,
 			       sizeof(format_page_default));
 
 			format_page = &lun->mode_pages.format_page[
 				CTL_PAGE_CURRENT];
 			scsi_ulto2b(lun->be_lun->blocksize,
 				    format_page->bytes_per_sector);
 
 			format_page = &lun->mode_pages.format_page[
 				CTL_PAGE_DEFAULT];
 			scsi_ulto2b(lun->be_lun->blocksize,
 				    format_page->bytes_per_sector);
 
 			format_page = &lun->mode_pages.format_page[
 				CTL_PAGE_SAVED];
 			scsi_ulto2b(lun->be_lun->blocksize,
 				    format_page->bytes_per_sector);
 
 			page_index->page_data =
 				(uint8_t *)lun->mode_pages.format_page;
 			break;
 		}
 		case SMS_RIGID_DISK_PAGE: {
 			struct scsi_rigid_disk_page *rigid_disk_page;
 			uint32_t sectors_per_cylinder;
 			uint64_t cylinders;
 #ifndef	__XSCALE__
 			int shift;
 #endif /* !__XSCALE__ */
 
 			if (page_index->subpage != SMS_SUBPAGE_PAGE_0)
 				panic("invalid subpage value %d",
 				      page_index->subpage);
 
 			/*
 			 * Rotation rate and sectors per track are set
 			 * above.  We calculate the cylinders here based on
 			 * capacity.  Due to the number of heads and
 			 * sectors per track we're using, smaller arrays
 			 * may turn out to have 0 cylinders.  Linux and
 			 * FreeBSD don't pay attention to these mode pages
 			 * to figure out capacity, but Solaris does.  It
 			 * seems to deal with 0 cylinders just fine, and
 			 * works out a fake geometry based on the capacity.
 			 */
 			memcpy(&lun->mode_pages.rigid_disk_page[
 			       CTL_PAGE_DEFAULT], &rigid_disk_page_default,
 			       sizeof(rigid_disk_page_default));
 			memcpy(&lun->mode_pages.rigid_disk_page[
 			       CTL_PAGE_CHANGEABLE],&rigid_disk_page_changeable,
 			       sizeof(rigid_disk_page_changeable));
 
 			sectors_per_cylinder = CTL_DEFAULT_SECTORS_PER_TRACK *
 				CTL_DEFAULT_HEADS;
 
 			/*
 			 * The divide method here will be more accurate,
 			 * probably, but results in floating point being
 			 * used in the kernel on i386 (__udivdi3()).  On the
 			 * XScale, though, __udivdi3() is implemented in
 			 * software.
 			 *
 			 * The shift method for cylinder calculation is
 			 * accurate if sectors_per_cylinder is a power of
 			 * 2.  Otherwise it might be slightly off -- you
 			 * might have a bit of a truncation problem.
 			 */
 #ifdef	__XSCALE__
 			cylinders = (lun->be_lun->maxlba + 1) /
 				sectors_per_cylinder;
 #else
 			for (shift = 31; shift > 0; shift--) {
 				if (sectors_per_cylinder & (1 << shift))
 					break;
 			}
 			cylinders = (lun->be_lun->maxlba + 1) >> shift;
 #endif
 
 			/*
 			 * We've basically got 3 bytes, or 24 bits for the
 			 * cylinder size in the mode page.  If we're over,
 			 * just round down to 2^24.
 			 */
 			if (cylinders > 0xffffff)
 				cylinders = 0xffffff;
 
 			rigid_disk_page = &lun->mode_pages.rigid_disk_page[
 				CTL_PAGE_DEFAULT];
 			scsi_ulto3b(cylinders, rigid_disk_page->cylinders);
 
 			if ((value = ctl_get_opt(&lun->be_lun->options,
 			    "rpm")) != NULL) {
 				scsi_ulto2b(strtol(value, NULL, 0),
 				     rigid_disk_page->rotation_rate);
 			}
 
 			memcpy(&lun->mode_pages.rigid_disk_page[CTL_PAGE_CURRENT],
 			       &lun->mode_pages.rigid_disk_page[CTL_PAGE_DEFAULT],
 			       sizeof(rigid_disk_page_default));
 			memcpy(&lun->mode_pages.rigid_disk_page[CTL_PAGE_SAVED],
 			       &lun->mode_pages.rigid_disk_page[CTL_PAGE_DEFAULT],
 			       sizeof(rigid_disk_page_default));
 
 			page_index->page_data =
 				(uint8_t *)lun->mode_pages.rigid_disk_page;
 			break;
 		}
 		case SMS_CACHING_PAGE: {
 			struct scsi_caching_page *caching_page;
 
 			if (page_index->subpage != SMS_SUBPAGE_PAGE_0)
 				panic("invalid subpage value %d",
 				      page_index->subpage);
 			memcpy(&lun->mode_pages.caching_page[CTL_PAGE_DEFAULT],
 			       &caching_page_default,
 			       sizeof(caching_page_default));
 			memcpy(&lun->mode_pages.caching_page[
 			       CTL_PAGE_CHANGEABLE], &caching_page_changeable,
 			       sizeof(caching_page_changeable));
 			memcpy(&lun->mode_pages.caching_page[CTL_PAGE_SAVED],
 			       &caching_page_default,
 			       sizeof(caching_page_default));
 			caching_page = &lun->mode_pages.caching_page[
 			    CTL_PAGE_SAVED];
 			value = ctl_get_opt(&lun->be_lun->options, "writecache");
 			if (value != NULL && strcmp(value, "off") == 0)
 				caching_page->flags1 &= ~SCP_WCE;
 			value = ctl_get_opt(&lun->be_lun->options, "readcache");
 			if (value != NULL && strcmp(value, "off") == 0)
 				caching_page->flags1 |= SCP_RCD;
 			memcpy(&lun->mode_pages.caching_page[CTL_PAGE_CURRENT],
 			       &lun->mode_pages.caching_page[CTL_PAGE_SAVED],
 			       sizeof(caching_page_default));
 			page_index->page_data =
 				(uint8_t *)lun->mode_pages.caching_page;
 			break;
 		}
 		case SMS_CONTROL_MODE_PAGE: {
 			struct scsi_control_page *control_page;
 
 			if (page_index->subpage != SMS_SUBPAGE_PAGE_0)
 				panic("invalid subpage value %d",
 				      page_index->subpage);
 
 			memcpy(&lun->mode_pages.control_page[CTL_PAGE_DEFAULT],
 			       &control_page_default,
 			       sizeof(control_page_default));
 			memcpy(&lun->mode_pages.control_page[
 			       CTL_PAGE_CHANGEABLE], &control_page_changeable,
 			       sizeof(control_page_changeable));
 			memcpy(&lun->mode_pages.control_page[CTL_PAGE_SAVED],
 			       &control_page_default,
 			       sizeof(control_page_default));
 			control_page = &lun->mode_pages.control_page[
 			    CTL_PAGE_SAVED];
 			value = ctl_get_opt(&lun->be_lun->options, "reordering");
 			if (value != NULL && strcmp(value, "unrestricted") == 0) {
 				control_page->queue_flags &= ~SCP_QUEUE_ALG_MASK;
 				control_page->queue_flags |= SCP_QUEUE_ALG_UNRESTRICTED;
 			}
 			memcpy(&lun->mode_pages.control_page[CTL_PAGE_CURRENT],
 			       &lun->mode_pages.control_page[CTL_PAGE_SAVED],
 			       sizeof(control_page_default));
 			page_index->page_data =
 				(uint8_t *)lun->mode_pages.control_page;
 			break;
 
 		}
 		case SMS_INFO_EXCEPTIONS_PAGE: {
 			switch (page_index->subpage) {
 			case SMS_SUBPAGE_PAGE_0:
 				memcpy(&lun->mode_pages.ie_page[CTL_PAGE_CURRENT],
 				       &ie_page_default,
 				       sizeof(ie_page_default));
 				memcpy(&lun->mode_pages.ie_page[
 				       CTL_PAGE_CHANGEABLE], &ie_page_changeable,
 				       sizeof(ie_page_changeable));
 				memcpy(&lun->mode_pages.ie_page[CTL_PAGE_DEFAULT],
 				       &ie_page_default,
 				       sizeof(ie_page_default));
 				memcpy(&lun->mode_pages.ie_page[CTL_PAGE_SAVED],
 				       &ie_page_default,
 				       sizeof(ie_page_default));
 				page_index->page_data =
 					(uint8_t *)lun->mode_pages.ie_page;
 				break;
 			case 0x02: {
 				struct ctl_logical_block_provisioning_page *page;
 
 				memcpy(&lun->mode_pages.lbp_page[CTL_PAGE_DEFAULT],
 				       &lbp_page_default,
 				       sizeof(lbp_page_default));
 				memcpy(&lun->mode_pages.lbp_page[
 				       CTL_PAGE_CHANGEABLE], &lbp_page_changeable,
 				       sizeof(lbp_page_changeable));
 				memcpy(&lun->mode_pages.lbp_page[CTL_PAGE_SAVED],
 				       &lbp_page_default,
 				       sizeof(lbp_page_default));
 				page = &lun->mode_pages.lbp_page[CTL_PAGE_SAVED];
 				value = ctl_get_opt(&lun->be_lun->options,
 				    "avail-threshold");
 				if (value != NULL &&
 				    ctl_expand_number(value, &ival) == 0) {
 					page->descr[0].flags |= SLBPPD_ENABLED |
 					    SLBPPD_ARMING_DEC;
 					if (lun->be_lun->blocksize)
 						ival /= lun->be_lun->blocksize;
 					else
 						ival /= 512;
 					scsi_ulto4b(ival >> CTL_LBP_EXPONENT,
 					    page->descr[0].count);
 				}
 				value = ctl_get_opt(&lun->be_lun->options,
 				    "used-threshold");
 				if (value != NULL &&
 				    ctl_expand_number(value, &ival) == 0) {
 					page->descr[1].flags |= SLBPPD_ENABLED |
 					    SLBPPD_ARMING_INC;
 					if (lun->be_lun->blocksize)
 						ival /= lun->be_lun->blocksize;
 					else
 						ival /= 512;
 					scsi_ulto4b(ival >> CTL_LBP_EXPONENT,
 					    page->descr[1].count);
 				}
 				value = ctl_get_opt(&lun->be_lun->options,
 				    "pool-avail-threshold");
 				if (value != NULL &&
 				    ctl_expand_number(value, &ival) == 0) {
 					page->descr[2].flags |= SLBPPD_ENABLED |
 					    SLBPPD_ARMING_DEC;
 					if (lun->be_lun->blocksize)
 						ival /= lun->be_lun->blocksize;
 					else
 						ival /= 512;
 					scsi_ulto4b(ival >> CTL_LBP_EXPONENT,
 					    page->descr[2].count);
 				}
 				value = ctl_get_opt(&lun->be_lun->options,
 				    "pool-used-threshold");
 				if (value != NULL &&
 				    ctl_expand_number(value, &ival) == 0) {
 					page->descr[3].flags |= SLBPPD_ENABLED |
 					    SLBPPD_ARMING_INC;
 					if (lun->be_lun->blocksize)
 						ival /= lun->be_lun->blocksize;
 					else
 						ival /= 512;
 					scsi_ulto4b(ival >> CTL_LBP_EXPONENT,
 					    page->descr[3].count);
 				}
 				memcpy(&lun->mode_pages.lbp_page[CTL_PAGE_CURRENT],
 				       &lun->mode_pages.lbp_page[CTL_PAGE_SAVED],
 				       sizeof(lbp_page_default));
 				page_index->page_data =
 					(uint8_t *)lun->mode_pages.lbp_page;
 			}}
 			break;
 		}
 		case SMS_VENDOR_SPECIFIC_PAGE:{
 			switch (page_index->subpage) {
 			case DBGCNF_SUBPAGE_CODE: {
 				struct copan_debugconf_subpage *current_page,
 							       *saved_page;
 
 				memcpy(&lun->mode_pages.debugconf_subpage[
 				       CTL_PAGE_CURRENT],
 				       &debugconf_page_default,
 				       sizeof(debugconf_page_default));
 				memcpy(&lun->mode_pages.debugconf_subpage[
 				       CTL_PAGE_CHANGEABLE],
 				       &debugconf_page_changeable,
 				       sizeof(debugconf_page_changeable));
 				memcpy(&lun->mode_pages.debugconf_subpage[
 				       CTL_PAGE_DEFAULT],
 				       &debugconf_page_default,
 				       sizeof(debugconf_page_default));
 				memcpy(&lun->mode_pages.debugconf_subpage[
 				       CTL_PAGE_SAVED],
 				       &debugconf_page_default,
 				       sizeof(debugconf_page_default));
 				page_index->page_data =
 					(uint8_t *)lun->mode_pages.debugconf_subpage;
 
 				current_page = (struct copan_debugconf_subpage *)
 					(page_index->page_data +
 					 (page_index->page_len *
 					  CTL_PAGE_CURRENT));
 				saved_page = (struct copan_debugconf_subpage *)
 					(page_index->page_data +
 					 (page_index->page_len *
 					  CTL_PAGE_SAVED));
 				break;
 			}
 			default:
 				panic("invalid subpage value %d",
 				      page_index->subpage);
 				break;
 			}
    			break;
 		}
 		default:
 			panic("invalid page value %d",
 			      page_index->page_code & SMPH_PC_MASK);
 			break;
     	}
 	}
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 ctl_init_log_page_index(struct ctl_lun *lun)
 {
 	struct ctl_page_index *page_index;
 	int i, j, k, prev;
 
 	memcpy(&lun->log_pages.index, log_page_index_template,
 	       sizeof(log_page_index_template));
 
 	prev = -1;
 	for (i = 0, j = 0, k = 0; i < CTL_NUM_LOG_PAGES; i++) {
 
 		page_index = &lun->log_pages.index[i];
 		/*
 		 * If this is a disk-only mode page, there's no point in
 		 * setting it up.  For some pages, we have to have some
 		 * basic information about the disk in order to calculate the
 		 * mode page data.
 		 */
 		if ((lun->be_lun->lun_type != T_DIRECT)
 		 && (page_index->page_flags & CTL_PAGE_FLAG_DISK_ONLY))
 			continue;
 
 		if (page_index->page_code == SLS_LOGICAL_BLOCK_PROVISIONING &&
 		     lun->backend->lun_attr == NULL)
 			continue;
 
 		if (page_index->page_code != prev) {
 			lun->log_pages.pages_page[j] = page_index->page_code;
 			prev = page_index->page_code;
 			j++;
 		}
 		lun->log_pages.subpages_page[k*2] = page_index->page_code;
 		lun->log_pages.subpages_page[k*2+1] = page_index->subpage;
 		k++;
 	}
 	lun->log_pages.index[0].page_data = &lun->log_pages.pages_page[0];
 	lun->log_pages.index[0].page_len = j;
 	lun->log_pages.index[1].page_data = &lun->log_pages.subpages_page[0];
 	lun->log_pages.index[1].page_len = k * 2;
 	lun->log_pages.index[2].page_data = &lun->log_pages.lbp_page[0];
 	lun->log_pages.index[2].page_len = 12*CTL_NUM_LBP_PARAMS;
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 hex2bin(const char *str, uint8_t *buf, int buf_size)
 {
 	int i;
 	u_char c;
 
 	memset(buf, 0, buf_size);
 	while (isspace(str[0]))
 		str++;
 	if (str[0] == '0' && (str[1] == 'x' || str[1] == 'X'))
 		str += 2;
 	buf_size *= 2;
 	for (i = 0; str[i] != 0 && i < buf_size; i++) {
 		c = str[i];
 		if (isdigit(c))
 			c -= '0';
 		else if (isalpha(c))
 			c -= isupper(c) ? 'A' - 10 : 'a' - 10;
 		else
 			break;
 		if (c >= 16)
 			break;
 		if ((i & 1) == 0)
 			buf[i / 2] |= (c << 4);
 		else
 			buf[i / 2] |= c;
 	}
 	return ((i + 1) / 2);
 }
 
 /*
  * LUN allocation.
  *
  * Requirements:
  * - caller allocates and zeros LUN storage, or passes in a NULL LUN if he
  *   wants us to allocate the LUN and he can block.
  * - ctl_softc is always set
  * - be_lun is set if the LUN has a backend (needed for disk LUNs)
  *
  * Returns 0 for success, non-zero (errno) for failure.
  */
 static int
 ctl_alloc_lun(struct ctl_softc *ctl_softc, struct ctl_lun *ctl_lun,
 	      struct ctl_be_lun *const be_lun, struct ctl_id target_id)
 {
 	struct ctl_lun *nlun, *lun;
 	struct ctl_port *port;
 	struct scsi_vpd_id_descriptor *desc;
 	struct scsi_vpd_id_t10 *t10id;
 	const char *eui, *naa, *scsiname, *vendor, *value;
 	int lun_number, i, lun_malloced;
 	int devidlen, idlen1, idlen2 = 0, len;
 
 	if (be_lun == NULL)
 		return (EINVAL);
 
 	/*
 	 * We currently only support Direct Access or Processor LUN types.
 	 */
 	switch (be_lun->lun_type) {
 	case T_DIRECT:
 		break;
 	case T_PROCESSOR:
 		break;
 	case T_SEQUENTIAL:
 	case T_CHANGER:
 	default:
 		be_lun->lun_config_status(be_lun->be_lun,
 					  CTL_LUN_CONFIG_FAILURE);
 		break;
 	}
 	if (ctl_lun == NULL) {
 		lun = malloc(sizeof(*lun), M_CTL, M_WAITOK);
 		lun_malloced = 1;
 	} else {
 		lun_malloced = 0;
 		lun = ctl_lun;
 	}
 
 	memset(lun, 0, sizeof(*lun));
 	if (lun_malloced)
 		lun->flags = CTL_LUN_MALLOCED;
 
 	/* Generate LUN ID. */
 	devidlen = max(CTL_DEVID_MIN_LEN,
 	    strnlen(be_lun->device_id, CTL_DEVID_LEN));
 	idlen1 = sizeof(*t10id) + devidlen;
 	len = sizeof(struct scsi_vpd_id_descriptor) + idlen1;
 	scsiname = ctl_get_opt(&be_lun->options, "scsiname");
 	if (scsiname != NULL) {
 		idlen2 = roundup2(strlen(scsiname) + 1, 4);
 		len += sizeof(struct scsi_vpd_id_descriptor) + idlen2;
 	}
 	eui = ctl_get_opt(&be_lun->options, "eui");
 	if (eui != NULL) {
 		len += sizeof(struct scsi_vpd_id_descriptor) + 16;
 	}
 	naa = ctl_get_opt(&be_lun->options, "naa");
 	if (naa != NULL) {
 		len += sizeof(struct scsi_vpd_id_descriptor) + 16;
 	}
 	lun->lun_devid = malloc(sizeof(struct ctl_devid) + len,
 	    M_CTL, M_WAITOK | M_ZERO);
 	desc = (struct scsi_vpd_id_descriptor *)lun->lun_devid->data;
 	desc->proto_codeset = SVPD_ID_CODESET_ASCII;
 	desc->id_type = SVPD_ID_PIV | SVPD_ID_ASSOC_LUN | SVPD_ID_TYPE_T10;
 	desc->length = idlen1;
 	t10id = (struct scsi_vpd_id_t10 *)&desc->identifier[0];
 	memset(t10id->vendor, ' ', sizeof(t10id->vendor));
 	if ((vendor = ctl_get_opt(&be_lun->options, "vendor")) == NULL) {
 		strncpy((char *)t10id->vendor, CTL_VENDOR, sizeof(t10id->vendor));
 	} else {
 		strncpy(t10id->vendor, vendor,
 		    min(sizeof(t10id->vendor), strlen(vendor)));
 	}
 	strncpy((char *)t10id->vendor_spec_id,
 	    (char *)be_lun->device_id, devidlen);
 	if (scsiname != NULL) {
 		desc = (struct scsi_vpd_id_descriptor *)(&desc->identifier[0] +
 		    desc->length);
 		desc->proto_codeset = SVPD_ID_CODESET_UTF8;
 		desc->id_type = SVPD_ID_PIV | SVPD_ID_ASSOC_LUN |
 		    SVPD_ID_TYPE_SCSI_NAME;
 		desc->length = idlen2;
 		strlcpy(desc->identifier, scsiname, idlen2);
 	}
 	if (eui != NULL) {
 		desc = (struct scsi_vpd_id_descriptor *)(&desc->identifier[0] +
 		    desc->length);
 		desc->proto_codeset = SVPD_ID_CODESET_BINARY;
 		desc->id_type = SVPD_ID_PIV | SVPD_ID_ASSOC_LUN |
 		    SVPD_ID_TYPE_EUI64;
 		desc->length = hex2bin(eui, desc->identifier, 16);
 		desc->length = desc->length > 12 ? 16 :
 		    (desc->length > 8 ? 12 : 8);
 		len -= 16 - desc->length;
 	}
 	if (naa != NULL) {
 		desc = (struct scsi_vpd_id_descriptor *)(&desc->identifier[0] +
 		    desc->length);
 		desc->proto_codeset = SVPD_ID_CODESET_BINARY;
 		desc->id_type = SVPD_ID_PIV | SVPD_ID_ASSOC_LUN |
 		    SVPD_ID_TYPE_NAA;
 		desc->length = hex2bin(naa, desc->identifier, 16);
 		desc->length = desc->length > 8 ? 16 : 8;
 		len -= 16 - desc->length;
 	}
 	lun->lun_devid->len = len;
 
 	mtx_lock(&ctl_softc->ctl_lock);
 	/*
 	 * See if the caller requested a particular LUN number.  If so, see
 	 * if it is available.  Otherwise, allocate the first available LUN.
 	 */
 	if (be_lun->flags & CTL_LUN_FLAG_ID_REQ) {
 		if ((be_lun->req_lun_id > (CTL_MAX_LUNS - 1))
 		 || (ctl_is_set(ctl_softc->ctl_lun_mask, be_lun->req_lun_id))) {
 			mtx_unlock(&ctl_softc->ctl_lock);
 			if (be_lun->req_lun_id > (CTL_MAX_LUNS - 1)) {
 				printf("ctl: requested LUN ID %d is higher "
 				       "than CTL_MAX_LUNS - 1 (%d)\n",
 				       be_lun->req_lun_id, CTL_MAX_LUNS - 1);
 			} else {
 				/*
 				 * XXX KDM return an error, or just assign
 				 * another LUN ID in this case??
 				 */
 				printf("ctl: requested LUN ID %d is already "
 				       "in use\n", be_lun->req_lun_id);
 			}
 			if (lun->flags & CTL_LUN_MALLOCED)
 				free(lun, M_CTL);
 			be_lun->lun_config_status(be_lun->be_lun,
 						  CTL_LUN_CONFIG_FAILURE);
 			return (ENOSPC);
 		}
 		lun_number = be_lun->req_lun_id;
 	} else {
 		lun_number = ctl_ffz(ctl_softc->ctl_lun_mask, CTL_MAX_LUNS);
 		if (lun_number == -1) {
 			mtx_unlock(&ctl_softc->ctl_lock);
 			printf("ctl: can't allocate LUN on target %ju, out of "
 			       "LUNs\n", (uintmax_t)target_id.id);
 			if (lun->flags & CTL_LUN_MALLOCED)
 				free(lun, M_CTL);
 			be_lun->lun_config_status(be_lun->be_lun,
 						  CTL_LUN_CONFIG_FAILURE);
 			return (ENOSPC);
 		}
 	}
 	ctl_set_mask(ctl_softc->ctl_lun_mask, lun_number);
 
 	mtx_init(&lun->lun_lock, "CTL LUN", NULL, MTX_DEF);
 	lun->target = target_id;
 	lun->lun = lun_number;
 	lun->be_lun = be_lun;
 	/*
 	 * The processor LUN is always enabled.  Disk LUNs come on line
 	 * disabled, and must be enabled by the backend.
 	 */
 	lun->flags |= CTL_LUN_DISABLED;
 	lun->backend = be_lun->be;
 	be_lun->ctl_lun = lun;
 	be_lun->lun_id = lun_number;
 	atomic_add_int(&be_lun->be->num_luns, 1);
 	if (be_lun->flags & CTL_LUN_FLAG_OFFLINE)
 		lun->flags |= CTL_LUN_OFFLINE;
 
 	if (be_lun->flags & CTL_LUN_FLAG_POWERED_OFF)
 		lun->flags |= CTL_LUN_STOPPED;
 
 	if (be_lun->flags & CTL_LUN_FLAG_INOPERABLE)
 		lun->flags |= CTL_LUN_INOPERABLE;
 
 	if (be_lun->flags & CTL_LUN_FLAG_PRIMARY)
 		lun->flags |= CTL_LUN_PRIMARY_SC;
 
 	value = ctl_get_opt(&be_lun->options, "readonly");
 	if (value != NULL && strcmp(value, "on") == 0)
 		lun->flags |= CTL_LUN_READONLY;
 
 	lun->serseq = CTL_LUN_SERSEQ_OFF;
 	if (be_lun->flags & CTL_LUN_FLAG_SERSEQ_READ)
 		lun->serseq = CTL_LUN_SERSEQ_READ;
 	value = ctl_get_opt(&be_lun->options, "serseq");
 	if (value != NULL && strcmp(value, "on") == 0)
 		lun->serseq = CTL_LUN_SERSEQ_ON;
 	else if (value != NULL && strcmp(value, "read") == 0)
 		lun->serseq = CTL_LUN_SERSEQ_READ;
 	else if (value != NULL && strcmp(value, "off") == 0)
 		lun->serseq = CTL_LUN_SERSEQ_OFF;
 
 	lun->ctl_softc = ctl_softc;
 	TAILQ_INIT(&lun->ooa_queue);
 	TAILQ_INIT(&lun->blocked_queue);
 	STAILQ_INIT(&lun->error_list);
 	ctl_tpc_lun_init(lun);
 
 	/*
 	 * Initialize the mode and log page index.
 	 */
 	ctl_init_page_index(lun);
 	ctl_init_log_page_index(lun);
 
 	/*
 	 * Now, before we insert this lun on the lun list, set the lun
 	 * inventory changed UA for all other luns.
 	 */
 	STAILQ_FOREACH(nlun, &ctl_softc->lun_list, links) {
 		mtx_lock(&nlun->lun_lock);
 		ctl_est_ua_all(nlun, -1, CTL_UA_LUN_CHANGE);
 		mtx_unlock(&nlun->lun_lock);
 	}
 
 	STAILQ_INSERT_TAIL(&ctl_softc->lun_list, lun, links);
 
 	ctl_softc->ctl_luns[lun_number] = lun;
 
 	ctl_softc->num_luns++;
 
 	/* Setup statistics gathering */
 	lun->stats.device_type = be_lun->lun_type;
 	lun->stats.lun_number = lun_number;
 	if (lun->stats.device_type == T_DIRECT)
 		lun->stats.blocksize = be_lun->blocksize;
 	else
 		lun->stats.flags = CTL_LUN_STATS_NO_BLOCKSIZE;
 	for (i = 0;i < CTL_MAX_PORTS;i++)
 		lun->stats.ports[i].targ_port = i;
 
 	mtx_unlock(&ctl_softc->ctl_lock);
 
 	lun->be_lun->lun_config_status(lun->be_lun->be_lun, CTL_LUN_CONFIG_OK);
 
 	/*
 	 * Run through each registered FETD and bring it online if it isn't
 	 * already.  Enable the target ID if it hasn't been enabled, and
 	 * enable this particular LUN.
 	 */
 	STAILQ_FOREACH(port, &ctl_softc->port_list, links) {
 		int retval;
 
 		retval = port->lun_enable(port->targ_lun_arg, target_id,lun_number);
 		if (retval != 0) {
 			printf("ctl_alloc_lun: FETD %s port %d returned error "
 			       "%d for lun_enable on target %ju lun %d\n",
 			       port->port_name, port->targ_port, retval,
 			       (uintmax_t)target_id.id, lun_number);
 		} else
 			port->status |= CTL_PORT_STATUS_LUN_ONLINE;
 	}
 	return (0);
 }
 
 /*
  * Delete a LUN.
  * Assumptions:
  * - LUN has already been marked invalid and any pending I/O has been taken
  *   care of.
  */
 static int
 ctl_free_lun(struct ctl_lun *lun)
 {
 	struct ctl_softc *softc;
 #if 0
 	struct ctl_port *port;
 #endif
 	struct ctl_lun *nlun;
 	int i;
 
 	softc = lun->ctl_softc;
 
 	mtx_assert(&softc->ctl_lock, MA_OWNED);
 
 	STAILQ_REMOVE(&softc->lun_list, lun, ctl_lun, links);
 
 	ctl_clear_mask(softc->ctl_lun_mask, lun->lun);
 
 	softc->ctl_luns[lun->lun] = NULL;
 
 	if (!TAILQ_EMPTY(&lun->ooa_queue))
 		panic("Freeing a LUN %p with outstanding I/O!!\n", lun);
 
 	softc->num_luns--;
 
 	/*
 	 * XXX KDM this scheme only works for a single target/multiple LUN
 	 * setup.  It needs to be revamped for a multiple target scheme.
 	 *
 	 * XXX KDM this results in port->lun_disable() getting called twice,
 	 * once when ctl_disable_lun() is called, and a second time here.
 	 * We really need to re-think the LUN disable semantics.  There
 	 * should probably be several steps/levels to LUN removal:
 	 *  - disable
 	 *  - invalidate
 	 *  - free
  	 *
 	 * Right now we only have a disable method when communicating to
 	 * the front end ports, at least for individual LUNs.
 	 */
 #if 0
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		int retval;
 
 		retval = port->lun_disable(port->targ_lun_arg, lun->target,
 					 lun->lun);
 		if (retval != 0) {
 			printf("ctl_free_lun: FETD %s port %d returned error "
 			       "%d for lun_disable on target %ju lun %jd\n",
 			       port->port_name, port->targ_port, retval,
 			       (uintmax_t)lun->target.id, (intmax_t)lun->lun);
 		}
 
 		if (STAILQ_FIRST(&softc->lun_list) == NULL) {
 			port->status &= ~CTL_PORT_STATUS_LUN_ONLINE;
 
 			retval = port->targ_disable(port->targ_lun_arg,lun->target);
 			if (retval != 0) {
 				printf("ctl_free_lun: FETD %s port %d "
 				       "returned error %d for targ_disable on "
 				       "target %ju\n", port->port_name,
 				       port->targ_port, retval,
 				       (uintmax_t)lun->target.id);
 			} else
 				port->status &= ~CTL_PORT_STATUS_TARG_ONLINE;
 
 			if ((port->status & CTL_PORT_STATUS_TARG_ONLINE) != 0)
 				continue;
 
 #if 0
 			port->port_offline(port->onoff_arg);
 			port->status &= ~CTL_PORT_STATUS_ONLINE;
 #endif
 		}
 	}
 #endif
 
 	/*
 	 * Tell the backend to free resources, if this LUN has a backend.
 	 */
 	atomic_subtract_int(&lun->be_lun->be->num_luns, 1);
 	lun->be_lun->lun_shutdown(lun->be_lun->be_lun);
 
 	ctl_tpc_lun_shutdown(lun);
 	mtx_destroy(&lun->lun_lock);
 	free(lun->lun_devid, M_CTL);
 	for (i = 0; i < CTL_MAX_PORTS; i++)
 		free(lun->pending_ua[i], M_CTL);
 	for (i = 0; i < 2 * CTL_MAX_PORTS; i++)
 		free(lun->pr_keys[i], M_CTL);
 	free(lun->write_buffer, M_CTL);
 	if (lun->flags & CTL_LUN_MALLOCED)
 		free(lun, M_CTL);
 
 	STAILQ_FOREACH(nlun, &softc->lun_list, links) {
 		mtx_lock(&nlun->lun_lock);
 		ctl_est_ua_all(nlun, -1, CTL_UA_LUN_CHANGE);
 		mtx_unlock(&nlun->lun_lock);
 	}
 
 	return (0);
 }
 
 static void
 ctl_create_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_softc *softc;
 
 	softc = control_softc;
 
 	/*
 	 * ctl_alloc_lun() should handle all potential failure cases.
 	 */
 	ctl_alloc_lun(softc, NULL, be_lun, softc->target);
 }
 
 int
 ctl_add_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_softc *softc = control_softc;
 
 	mtx_lock(&softc->ctl_lock);
 	STAILQ_INSERT_TAIL(&softc->pending_lun_queue, be_lun, links);
 	mtx_unlock(&softc->ctl_lock);
 	wakeup(&softc->pending_lun_queue);
 
 	return (0);
 }
 
 int
 ctl_enable_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_softc *softc;
 	struct ctl_port *port, *nport;
 	struct ctl_lun *lun;
 	int retval;
 
 	lun = (struct ctl_lun *)be_lun->ctl_lun;
 	softc = lun->ctl_softc;
 
 	mtx_lock(&softc->ctl_lock);
 	mtx_lock(&lun->lun_lock);
 	if ((lun->flags & CTL_LUN_DISABLED) == 0) {
 		/*
 		 * eh?  Why did we get called if the LUN is already
 		 * enabled?
 		 */
 		mtx_unlock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 		return (0);
 	}
 	lun->flags &= ~CTL_LUN_DISABLED;
 	mtx_unlock(&lun->lun_lock);
 
 	for (port = STAILQ_FIRST(&softc->port_list); port != NULL; port = nport) {
 		nport = STAILQ_NEXT(port, links);
 
 		/*
 		 * Drop the lock while we call the FETD's enable routine.
 		 * This can lead to a callback into CTL (at least in the
 		 * case of the internal initiator frontend.
 		 */
 		mtx_unlock(&softc->ctl_lock);
 		retval = port->lun_enable(port->targ_lun_arg, lun->target,lun->lun);
 		mtx_lock(&softc->ctl_lock);
 		if (retval != 0) {
 			printf("%s: FETD %s port %d returned error "
 			       "%d for lun_enable on target %ju lun %jd\n",
 			       __func__, port->port_name, port->targ_port, retval,
 			       (uintmax_t)lun->target.id, (intmax_t)lun->lun);
 		}
 #if 0
 		 else {
             /* NOTE:  TODO:  why does lun enable affect port status? */
 			port->status |= CTL_PORT_STATUS_LUN_ONLINE;
 		}
 #endif
 	}
 
 	mtx_unlock(&softc->ctl_lock);
 
 	return (0);
 }
 
 int
 ctl_disable_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_softc *softc;
 	struct ctl_port *port;
 	struct ctl_lun *lun;
 	int retval;
 
 	lun = (struct ctl_lun *)be_lun->ctl_lun;
 	softc = lun->ctl_softc;
 
 	mtx_lock(&softc->ctl_lock);
 	mtx_lock(&lun->lun_lock);
 	if (lun->flags & CTL_LUN_DISABLED) {
 		mtx_unlock(&lun->lun_lock);
 		mtx_unlock(&softc->ctl_lock);
 		return (0);
 	}
 	lun->flags |= CTL_LUN_DISABLED;
 	mtx_unlock(&lun->lun_lock);
 
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		mtx_unlock(&softc->ctl_lock);
 		/*
 		 * Drop the lock before we call the frontend's disable
 		 * routine, to avoid lock order reversals.
 		 *
 		 * XXX KDM what happens if the frontend list changes while
 		 * we're traversing it?  It's unlikely, but should be handled.
 		 */
 		retval = port->lun_disable(port->targ_lun_arg, lun->target,
 					 lun->lun);
 		mtx_lock(&softc->ctl_lock);
 		if (retval != 0) {
 			printf("ctl_alloc_lun: FETD %s port %d returned error "
 			       "%d for lun_disable on target %ju lun %jd\n",
 			       port->port_name, port->targ_port, retval,
 			       (uintmax_t)lun->target.id, (intmax_t)lun->lun);
 		}
 	}
 
 	mtx_unlock(&softc->ctl_lock);
 
 	return (0);
 }
 
 int
 ctl_start_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	lun->flags &= ~CTL_LUN_STOPPED;
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 int
 ctl_stop_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	lun->flags |= CTL_LUN_STOPPED;
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 int
 ctl_lun_offline(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	lun->flags |= CTL_LUN_OFFLINE;
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 int
 ctl_lun_online(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	lun->flags &= ~CTL_LUN_OFFLINE;
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 int
 ctl_invalidate_lun(struct ctl_be_lun *be_lun)
 {
 	struct ctl_softc *softc;
 	struct ctl_lun *lun;
 
 	lun = (struct ctl_lun *)be_lun->ctl_lun;
 	softc = lun->ctl_softc;
 
 	mtx_lock(&lun->lun_lock);
 
 	/*
 	 * The LUN needs to be disabled before it can be marked invalid.
 	 */
 	if ((lun->flags & CTL_LUN_DISABLED) == 0) {
 		mtx_unlock(&lun->lun_lock);
 		return (-1);
 	}
 	/*
 	 * Mark the LUN invalid.
 	 */
 	lun->flags |= CTL_LUN_INVALID;
 
 	/*
 	 * If there is nothing in the OOA queue, go ahead and free the LUN.
 	 * If we have something in the OOA queue, we'll free it when the
 	 * last I/O completes.
 	 */
 	if (TAILQ_EMPTY(&lun->ooa_queue)) {
 		mtx_unlock(&lun->lun_lock);
 		mtx_lock(&softc->ctl_lock);
 		ctl_free_lun(lun);
 		mtx_unlock(&softc->ctl_lock);
 	} else
 		mtx_unlock(&lun->lun_lock);
 
 	return (0);
 }
 
 int
 ctl_lun_inoperable(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	lun->flags |= CTL_LUN_INOPERABLE;
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 int
 ctl_lun_operable(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	lun->flags &= ~CTL_LUN_INOPERABLE;
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 void
 ctl_lun_capacity_changed(struct ctl_be_lun *be_lun)
 {
 	struct ctl_lun *lun = (struct ctl_lun *)be_lun->ctl_lun;
 
 	mtx_lock(&lun->lun_lock);
 	ctl_est_ua_all(lun, -1, CTL_UA_CAPACITY_CHANGED);
 	mtx_unlock(&lun->lun_lock);
 }
 
 /*
  * Backend "memory move is complete" callback for requests that never
  * make it down to say RAIDCore's configuration code.
  */
 int
 ctl_config_move_done(union ctl_io *io)
 {
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_config_move_done\n"));
 	KASSERT(io->io_hdr.io_type == CTL_IO_SCSI,
 	    ("Config I/O type isn't CTL_IO_SCSI (%d)!", io->io_hdr.io_type));
 
 	if ((io->io_hdr.port_status != 0) &&
 	    ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_STATUS_NONE ||
 	     (io->io_hdr.status & CTL_STATUS_MASK) == CTL_SUCCESS)) {
 		/*
 		 * For hardware error sense keys, the sense key
 		 * specific value is defined to be a retry count,
 		 * but we use it to pass back an internal FETD
 		 * error code.  XXX KDM  Hopefully the FETD is only
 		 * using 16 bits for an error code, since that's
 		 * all the space we have in the sks field.
 		 */
 		ctl_set_internal_failure(&io->scsiio,
 					 /*sks_valid*/ 1,
 					 /*retry_count*/
 					 io->io_hdr.port_status);
 	}
 
 	if (((io->io_hdr.flags & CTL_FLAG_DATA_MASK) == CTL_FLAG_DATA_IN) ||
 	    ((io->io_hdr.status & CTL_STATUS_MASK) != CTL_STATUS_NONE &&
 	     (io->io_hdr.status & CTL_STATUS_MASK) != CTL_SUCCESS) ||
 	    ((io->io_hdr.flags & CTL_FLAG_ABORT) != 0)) {
 		/*
 		 * XXX KDM just assuming a single pointer here, and not a
 		 * S/G list.  If we start using S/G lists for config data,
 		 * we'll need to know how to clean them up here as well.
 		 */
 		if (io->io_hdr.flags & CTL_FLAG_ALLOCATED)
 			free(io->scsiio.kern_data_ptr, M_CTL);
 		ctl_done(io);
 		retval = CTL_RETVAL_COMPLETE;
 	} else {
 		/*
 		 * XXX KDM now we need to continue data movement.  Some
 		 * options:
 		 * - call ctl_scsiio() again?  We don't do this for data
 		 *   writes, because for those at least we know ahead of
 		 *   time where the write will go and how long it is.  For
 		 *   config writes, though, that information is largely
 		 *   contained within the write itself, thus we need to
 		 *   parse out the data again.
 		 *
 		 * - Call some other function once the data is in?
 		 */
 		if (ctl_debug & CTL_DEBUG_CDB_DATA)
 			ctl_data_print(io);
 
 		/*
 		 * XXX KDM call ctl_scsiio() again for now, and check flag
 		 * bits to see whether we're allocated or not.
 		 */
 		retval = ctl_scsiio(&io->scsiio);
 	}
 	return (retval);
 }
 
 /*
  * This gets called by a backend driver when it is done with a
  * data_submit method.
  */
 void
 ctl_data_submit_done(union ctl_io *io)
 {
 	/*
 	 * If the IO_CONT flag is set, we need to call the supplied
 	 * function to continue processing the I/O, instead of completing
 	 * the I/O just yet.
 	 *
 	 * If there is an error, though, we don't want to keep processing.
 	 * Instead, just send status back to the initiator.
 	 */
 	if ((io->io_hdr.flags & CTL_FLAG_IO_CONT) &&
 	    (io->io_hdr.flags & CTL_FLAG_ABORT) == 0 &&
 	    ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_STATUS_NONE ||
 	     (io->io_hdr.status & CTL_STATUS_MASK) == CTL_SUCCESS)) {
 		io->scsiio.io_cont(io);
 		return;
 	}
 	ctl_done(io);
 }
 
 /*
  * This gets called by a backend driver when it is done with a
  * configuration write.
  */
 void
 ctl_config_write_done(union ctl_io *io)
 {
 	uint8_t *buf;
 
 	/*
 	 * If the IO_CONT flag is set, we need to call the supplied
 	 * function to continue processing the I/O, instead of completing
 	 * the I/O just yet.
 	 *
 	 * If there is an error, though, we don't want to keep processing.
 	 * Instead, just send status back to the initiator.
 	 */
 	if ((io->io_hdr.flags & CTL_FLAG_IO_CONT) &&
 	    (io->io_hdr.flags & CTL_FLAG_ABORT) == 0 &&
 	    ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_STATUS_NONE ||
 	     (io->io_hdr.status & CTL_STATUS_MASK) == CTL_SUCCESS)) {
 		io->scsiio.io_cont(io);
 		return;
 	}
 	/*
 	 * Since a configuration write can be done for commands that actually
 	 * have data allocated, like write buffer, and commands that have
 	 * no data, like start/stop unit, we need to check here.
 	 */
 	if (io->io_hdr.flags & CTL_FLAG_ALLOCATED)
 		buf = io->scsiio.kern_data_ptr;
 	else
 		buf = NULL;
 	ctl_done(io);
 	if (buf)
 		free(buf, M_CTL);
 }
 
 void
 ctl_config_read_done(union ctl_io *io)
 {
 	uint8_t *buf;
 
 	/*
 	 * If there is some error -- we are done, skip data transfer.
 	 */
 	if ((io->io_hdr.flags & CTL_FLAG_ABORT) != 0 ||
 	    ((io->io_hdr.status & CTL_STATUS_MASK) != CTL_STATUS_NONE &&
 	     (io->io_hdr.status & CTL_STATUS_MASK) != CTL_SUCCESS)) {
 		if (io->io_hdr.flags & CTL_FLAG_ALLOCATED)
 			buf = io->scsiio.kern_data_ptr;
 		else
 			buf = NULL;
 		ctl_done(io);
 		if (buf)
 			free(buf, M_CTL);
 		return;
 	}
 
 	/*
 	 * If the IO_CONT flag is set, we need to call the supplied
 	 * function to continue processing the I/O, instead of completing
 	 * the I/O just yet.
 	 */
 	if (io->io_hdr.flags & CTL_FLAG_IO_CONT) {
 		io->scsiio.io_cont(io);
 		return;
 	}
 
 	ctl_datamove(io);
 }
 
 /*
  * SCSI release command.
  */
 int
 ctl_scsi_release(struct ctl_scsiio *ctsio)
 {
 	int length, longid, thirdparty_id, resv_id;
 	struct ctl_lun *lun;
 	uint32_t residx;
 
 	length = 0;
 	resv_id = 0;
 
 	CTL_DEBUG_PRINT(("ctl_scsi_release\n"));
 
 	residx = ctl_get_resindex(&ctsio->io_hdr.nexus);
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	switch (ctsio->cdb[0]) {
 	case RELEASE_10: {
 		struct scsi_release_10 *cdb;
 
 		cdb = (struct scsi_release_10 *)ctsio->cdb;
 
 		if (cdb->byte2 & SR10_LONGID)
 			longid = 1;
 		else
 			thirdparty_id = cdb->thirdparty_id;
 
 		resv_id = cdb->resv_id;
 		length = scsi_2btoul(cdb->length);
 		break;
 	}
 	}
 
 
 	/*
 	 * XXX KDM right now, we only support LUN reservation.  We don't
 	 * support 3rd party reservations, or extent reservations, which
 	 * might actually need the parameter list.  If we've gotten this
 	 * far, we've got a LUN reservation.  Anything else got kicked out
 	 * above.  So, according to SPC, ignore the length.
 	 */
 	length = 0;
 
 	if (((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0)
 	 && (length > 0)) {
 		ctsio->kern_data_ptr = malloc(length, M_CTL, M_WAITOK);
 		ctsio->kern_data_len = length;
 		ctsio->kern_total_len = length;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	if (length > 0)
 		thirdparty_id = scsi_8btou64(ctsio->kern_data_ptr);
 
 	mtx_lock(&lun->lun_lock);
 
 	/*
 	 * According to SPC, it is not an error for an intiator to attempt
 	 * to release a reservation on a LUN that isn't reserved, or that
 	 * is reserved by another initiator.  The reservation can only be
 	 * released, though, by the initiator who made it or by one of
 	 * several reset type events.
 	 */
 	if ((lun->flags & CTL_LUN_RESERVED) && (lun->res_idx == residx))
 			lun->flags &= ~CTL_LUN_RESERVED;
 
 	mtx_unlock(&lun->lun_lock);
 
 	if (ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) {
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctsio->io_hdr.flags &= ~CTL_FLAG_ALLOCATED;
 	}
 
 	ctl_set_success(ctsio);
 	ctl_done((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_scsi_reserve(struct ctl_scsiio *ctsio)
 {
 	int extent, thirdparty, longid;
 	int resv_id, length;
 	uint64_t thirdparty_id;
 	struct ctl_lun *lun;
 	uint32_t residx;
 
 	extent = 0;
 	thirdparty = 0;
 	longid = 0;
 	resv_id = 0;
 	length = 0;
 	thirdparty_id = 0;
 
 	CTL_DEBUG_PRINT(("ctl_reserve\n"));
 
 	residx = ctl_get_resindex(&ctsio->io_hdr.nexus);
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	switch (ctsio->cdb[0]) {
 	case RESERVE_10: {
 		struct scsi_reserve_10 *cdb;
 
 		cdb = (struct scsi_reserve_10 *)ctsio->cdb;
 
 		if (cdb->byte2 & SR10_LONGID)
 			longid = 1;
 		else
 			thirdparty_id = cdb->thirdparty_id;
 
 		resv_id = cdb->resv_id;
 		length = scsi_2btoul(cdb->length);
 		break;
 	}
 	}
 
 	/*
 	 * XXX KDM right now, we only support LUN reservation.  We don't
 	 * support 3rd party reservations, or extent reservations, which
 	 * might actually need the parameter list.  If we've gotten this
 	 * far, we've got a LUN reservation.  Anything else got kicked out
 	 * above.  So, according to SPC, ignore the length.
 	 */
 	length = 0;
 
 	if (((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0)
 	 && (length > 0)) {
 		ctsio->kern_data_ptr = malloc(length, M_CTL, M_WAITOK);
 		ctsio->kern_data_len = length;
 		ctsio->kern_total_len = length;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	if (length > 0)
 		thirdparty_id = scsi_8btou64(ctsio->kern_data_ptr);
 
 	mtx_lock(&lun->lun_lock);
 	if ((lun->flags & CTL_LUN_RESERVED) && (lun->res_idx != residx)) {
 		ctl_set_reservation_conflict(ctsio);
 		goto bailout;
 	}
 
 	lun->flags |= CTL_LUN_RESERVED;
 	lun->res_idx = residx;
 
 	ctl_set_success(ctsio);
 
 bailout:
 	mtx_unlock(&lun->lun_lock);
 
 	if (ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) {
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctsio->io_hdr.flags &= ~CTL_FLAG_ALLOCATED;
 	}
 
 	ctl_done((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_start_stop(struct ctl_scsiio *ctsio)
 {
 	struct scsi_start_stop_unit *cdb;
 	struct ctl_lun *lun;
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_start_stop\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	retval = 0;
 
 	cdb = (struct scsi_start_stop_unit *)ctsio->cdb;
 
 	/*
 	 * XXX KDM
 	 * We don't support the immediate bit on a stop unit.  In order to
 	 * do that, we would need to code up a way to know that a stop is
 	 * pending, and hold off any new commands until it completes, one
 	 * way or another.  Then we could accept or reject those commands
 	 * depending on its status.  We would almost need to do the reverse
 	 * of what we do below for an immediate start -- return the copy of
 	 * the ctl_io to the FETD with status to send to the host (and to
 	 * free the copy!) and then free the original I/O once the stop
 	 * actually completes.  That way, the OOA queue mechanism can work
 	 * to block commands that shouldn't proceed.  Another alternative
 	 * would be to put the copy in the queue in place of the original,
 	 * and return the original back to the caller.  That could be
 	 * slightly safer..
 	 */
 	if ((cdb->byte2 & SSS_IMMED)
 	 && ((cdb->how & SSS_START) == 0)) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 1,
 				      /*bit_valid*/ 1,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	if ((lun->flags & CTL_LUN_PR_RESERVED)
 	 && ((cdb->how & SSS_START)==0)) {
 		uint32_t residx;
 
 		residx = ctl_get_resindex(&ctsio->io_hdr.nexus);
 		if (ctl_get_prkey(lun, residx) == 0
 		 || (lun->pr_res_idx!=residx && lun->res_type < 4)) {
 
 			ctl_set_reservation_conflict(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 	}
 
 	/*
 	 * If there is no backend on this device, we can't start or stop
 	 * it.  In theory we shouldn't get any start/stop commands in the
 	 * first place at this level if the LUN doesn't have a backend.
 	 * That should get stopped by the command decode code.
 	 */
 	if (lun->backend == NULL) {
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * XXX KDM Copan-specific offline behavior.
 	 * Figure out a reasonable way to port this?
 	 */
 #ifdef NEEDTOPORT
 	mtx_lock(&lun->lun_lock);
 
 	if (((cdb->byte2 & SSS_ONOFFLINE) == 0)
 	 && (lun->flags & CTL_LUN_OFFLINE)) {
 		/*
 		 * If the LUN is offline, and the on/offline bit isn't set,
 		 * reject the start or stop.  Otherwise, let it through.
 		 */
 		mtx_unlock(&lun->lun_lock);
 		ctl_set_lun_not_ready(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 	} else {
 		mtx_unlock(&lun->lun_lock);
 #endif /* NEEDTOPORT */
 		/*
 		 * This could be a start or a stop when we're online,
 		 * or a stop/offline or start/online.  A start or stop when
 		 * we're offline is covered in the case above.
 		 */
 		/*
 		 * In the non-immediate case, we send the request to
 		 * the backend and return status to the user when
 		 * it is done.
 		 *
 		 * In the immediate case, we allocate a new ctl_io
 		 * to hold a copy of the request, and send that to
 		 * the backend.  We then set good status on the
 		 * user's request and return it immediately.
 		 */
 		if (cdb->byte2 & SSS_IMMED) {
 			union ctl_io *new_io;
 
 			new_io = ctl_alloc_io(ctsio->io_hdr.pool);
 			ctl_copy_io((union ctl_io *)ctsio, new_io);
 			retval = lun->backend->config_write(new_io);
 			ctl_set_success(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 		} else {
 			retval = lun->backend->config_write(
 				(union ctl_io *)ctsio);
 		}
 #ifdef NEEDTOPORT
 	}
 #endif
 	return (retval);
 }
 
 /*
  * We support the SYNCHRONIZE CACHE command (10 and 16 byte versions), but
  * we don't really do anything with the LBA and length fields if the user
  * passes them in.  Instead we'll just flush out the cache for the entire
  * LUN.
  */
 int
 ctl_sync_cache(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	uint64_t starting_lba;
 	uint32_t block_count;
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_sync_cache\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	softc = lun->ctl_softc;
 	retval = 0;
 
 	switch (ctsio->cdb[0]) {
 	case SYNCHRONIZE_CACHE: {
 		struct scsi_sync_cache *cdb;
 		cdb = (struct scsi_sync_cache *)ctsio->cdb;
 
 		starting_lba = scsi_4btoul(cdb->begin_lba);
 		block_count = scsi_2btoul(cdb->lb_count);
 		break;
 	}
 	case SYNCHRONIZE_CACHE_16: {
 		struct scsi_sync_cache_16 *cdb;
 		cdb = (struct scsi_sync_cache_16 *)ctsio->cdb;
 
 		starting_lba = scsi_8btou64(cdb->begin_lba);
 		block_count = scsi_4btoul(cdb->lb_count);
 		break;
 	}
 	default:
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		goto bailout;
 		break; /* NOTREACHED */
 	}
 
 	/*
 	 * We check the LBA and length, but don't do anything with them.
 	 * A SYNCHRONIZE CACHE will cause the entire cache for this lun to
 	 * get flushed.  This check will just help satisfy anyone who wants
 	 * to see an error for an out of range LBA.
 	 */
 	if ((starting_lba + block_count) > (lun->be_lun->maxlba + 1)) {
 		ctl_set_lba_out_of_range(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		goto bailout;
 	}
 
 	/*
 	 * If this LUN has no backend, we can't flush the cache anyway.
 	 */
 	if (lun->backend == NULL) {
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		goto bailout;
 	}
 
 	/*
 	 * Check to see whether we're configured to send the SYNCHRONIZE
 	 * CACHE command directly to the back end.
 	 */
 	mtx_lock(&lun->lun_lock);
 	if ((softc->flags & CTL_FLAG_REAL_SYNC)
 	 && (++(lun->sync_count) >= lun->sync_interval)) {
 		lun->sync_count = 0;
 		mtx_unlock(&lun->lun_lock);
 		retval = lun->backend->config_write((union ctl_io *)ctsio);
 	} else {
 		mtx_unlock(&lun->lun_lock);
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 	}
 
 bailout:
 
 	return (retval);
 }
 
 int
 ctl_format(struct ctl_scsiio *ctsio)
 {
 	struct scsi_format *cdb;
 	struct ctl_lun *lun;
 	int length, defect_list_len;
 
 	CTL_DEBUG_PRINT(("ctl_format\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	cdb = (struct scsi_format *)ctsio->cdb;
 
 	length = 0;
 	if (cdb->byte2 & SF_FMTDATA) {
 		if (cdb->byte2 & SF_LONGLIST)
 			length = sizeof(struct scsi_format_header_long);
 		else
 			length = sizeof(struct scsi_format_header_short);
 	}
 
 	if (((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0)
 	 && (length > 0)) {
 		ctsio->kern_data_ptr = malloc(length, M_CTL, M_WAITOK);
 		ctsio->kern_data_len = length;
 		ctsio->kern_total_len = length;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	defect_list_len = 0;
 
 	if (cdb->byte2 & SF_FMTDATA) {
 		if (cdb->byte2 & SF_LONGLIST) {
 			struct scsi_format_header_long *header;
 
 			header = (struct scsi_format_header_long *)
 				ctsio->kern_data_ptr;
 
 			defect_list_len = scsi_4btoul(header->defect_list_len);
 			if (defect_list_len != 0) {
 				ctl_set_invalid_field(ctsio,
 						      /*sks_valid*/ 1,
 						      /*command*/ 0,
 						      /*field*/ 2,
 						      /*bit_valid*/ 0,
 						      /*bit*/ 0);
 				goto bailout;
 			}
 		} else {
 			struct scsi_format_header_short *header;
 
 			header = (struct scsi_format_header_short *)
 				ctsio->kern_data_ptr;
 
 			defect_list_len = scsi_2btoul(header->defect_list_len);
 			if (defect_list_len != 0) {
 				ctl_set_invalid_field(ctsio,
 						      /*sks_valid*/ 1,
 						      /*command*/ 0,
 						      /*field*/ 2,
 						      /*bit_valid*/ 0,
 						      /*bit*/ 0);
 				goto bailout;
 			}
 		}
 	}
 
 	/*
 	 * The format command will clear out the "Medium format corrupted"
 	 * status if set by the configuration code.  That status is really
 	 * just a way to notify the host that we have lost the media, and
 	 * get them to issue a command that will basically make them think
 	 * they're blowing away the media.
 	 */
 	mtx_lock(&lun->lun_lock);
 	lun->flags &= ~CTL_LUN_INOPERABLE;
 	mtx_unlock(&lun->lun_lock);
 
 	ctl_set_success(ctsio);
 bailout:
 
 	if (ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) {
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctsio->io_hdr.flags &= ~CTL_FLAG_ALLOCATED;
 	}
 
 	ctl_done((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_read_buffer(struct ctl_scsiio *ctsio)
 {
 	struct scsi_read_buffer *cdb;
 	struct ctl_lun *lun;
 	int buffer_offset, len;
 	static uint8_t descr[4];
 	static uint8_t echo_descr[4] = { 0 };
 
 	CTL_DEBUG_PRINT(("ctl_read_buffer\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_read_buffer *)ctsio->cdb;
 
 	if ((cdb->byte2 & RWB_MODE) != RWB_MODE_DATA &&
 	    (cdb->byte2 & RWB_MODE) != RWB_MODE_ECHO_DESCR &&
 	    (cdb->byte2 & RWB_MODE) != RWB_MODE_DESCR) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 1,
 				      /*bit_valid*/ 1,
 				      /*bit*/ 4);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	len = scsi_3btoul(cdb->length);
 	buffer_offset = scsi_3btoul(cdb->offset);
 
 	if (buffer_offset + len > CTL_WRITE_BUFFER_SIZE) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 6,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	if ((cdb->byte2 & RWB_MODE) == RWB_MODE_DESCR) {
 		descr[0] = 0;
 		scsi_ulto3b(CTL_WRITE_BUFFER_SIZE, &descr[1]);
 		ctsio->kern_data_ptr = descr;
 		len = min(len, sizeof(descr));
 	} else if ((cdb->byte2 & RWB_MODE) == RWB_MODE_ECHO_DESCR) {
 		ctsio->kern_data_ptr = echo_descr;
 		len = min(len, sizeof(echo_descr));
 	} else {
 		if (lun->write_buffer == NULL) {
 			lun->write_buffer = malloc(CTL_WRITE_BUFFER_SIZE,
 			    M_CTL, M_WAITOK);
 		}
 		ctsio->kern_data_ptr = lun->write_buffer + buffer_offset;
 	}
 	ctsio->kern_data_len = len;
 	ctsio->kern_total_len = len;
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 	ctl_set_success(ctsio);
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_write_buffer(struct ctl_scsiio *ctsio)
 {
 	struct scsi_write_buffer *cdb;
 	struct ctl_lun *lun;
 	int buffer_offset, len;
 
 	CTL_DEBUG_PRINT(("ctl_write_buffer\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_write_buffer *)ctsio->cdb;
 
 	if ((cdb->byte2 & RWB_MODE) != RWB_MODE_DATA) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 1,
 				      /*bit_valid*/ 1,
 				      /*bit*/ 4);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	len = scsi_3btoul(cdb->length);
 	buffer_offset = scsi_3btoul(cdb->offset);
 
 	if (buffer_offset + len > CTL_WRITE_BUFFER_SIZE) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 6,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * If we've got a kernel request that hasn't been malloced yet,
 	 * malloc it and tell the caller the data buffer is here.
 	 */
 	if ((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0) {
 		if (lun->write_buffer == NULL) {
 			lun->write_buffer = malloc(CTL_WRITE_BUFFER_SIZE,
 			    M_CTL, M_WAITOK);
 		}
 		ctsio->kern_data_ptr = lun->write_buffer + buffer_offset;
 		ctsio->kern_data_len = len;
 		ctsio->kern_total_len = len;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	ctl_set_success(ctsio);
 	ctl_done((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_write_same(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct ctl_lba_len_flags *lbalen;
 	uint64_t lba;
 	uint32_t num_blocks;
 	int len, retval;
 	uint8_t byte2;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	CTL_DEBUG_PRINT(("ctl_write_same\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	switch (ctsio->cdb[0]) {
 	case WRITE_SAME_10: {
 		struct scsi_write_same_10 *cdb;
 
 		cdb = (struct scsi_write_same_10 *)ctsio->cdb;
 
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_2btoul(cdb->length);
 		byte2 = cdb->byte2;
 		break;
 	}
 	case WRITE_SAME_16: {
 		struct scsi_write_same_16 *cdb;
 
 		cdb = (struct scsi_write_same_16 *)ctsio->cdb;
 
 		lba = scsi_8btou64(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		byte2 = cdb->byte2;
 		break;
 	}
 	default:
 		/*
 		 * We got a command we don't support.  This shouldn't
 		 * happen, commands should be filtered out above us.
 		 */
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 		break; /* NOTREACHED */
 	}
 
 	/* NDOB and ANCHOR flags can be used only together with UNMAP */
 	if ((byte2 & SWS_UNMAP) == 0 &&
 	    (byte2 & (SWS_NDOB | SWS_ANCHOR)) != 0) {
 		ctl_set_invalid_field(ctsio, /*sks_valid*/ 1,
 		    /*command*/ 1, /*field*/ 1, /*bit_valid*/ 1, /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * The first check is to make sure we're in bounds, the second
 	 * check is to catch wrap-around problems.  If the lba + num blocks
 	 * is less than the lba, then we've wrapped around and the block
 	 * range is invalid anyway.
 	 */
 	if (((lba + num_blocks) > (lun->be_lun->maxlba + 1))
 	 || ((lba + num_blocks) < lba)) {
 		ctl_set_lba_out_of_range(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/* Zero number of blocks means "to the last logical block" */
 	if (num_blocks == 0) {
 		if ((lun->be_lun->maxlba + 1) - lba > UINT32_MAX) {
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 0,
 					      /*command*/ 1,
 					      /*field*/ 0,
 					      /*bit_valid*/ 0,
 					      /*bit*/ 0);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		num_blocks = (lun->be_lun->maxlba + 1) - lba;
 	}
 
 	len = lun->be_lun->blocksize;
 
 	/*
 	 * If we've got a kernel request that hasn't been malloced yet,
 	 * malloc it and tell the caller the data buffer is here.
 	 */
 	if ((byte2 & SWS_NDOB) == 0 &&
 	    (ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0) {
 		ctsio->kern_data_ptr = malloc(len, M_CTL, M_WAITOK);;
 		ctsio->kern_data_len = len;
 		ctsio->kern_total_len = len;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	lbalen = (struct ctl_lba_len_flags *)&ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	lbalen->lba = lba;
 	lbalen->len = num_blocks;
 	lbalen->flags = byte2;
 	retval = lun->backend->config_write((union ctl_io *)ctsio);
 
 	return (retval);
 }
 
 int
 ctl_unmap(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct scsi_unmap *cdb;
 	struct ctl_ptr_len_flags *ptrlen;
 	struct scsi_unmap_header *hdr;
 	struct scsi_unmap_desc *buf, *end, *endnz, *range;
 	uint64_t lba;
 	uint32_t num_blocks;
 	int len, retval;
 	uint8_t byte2;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	CTL_DEBUG_PRINT(("ctl_unmap\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_unmap *)ctsio->cdb;
 
 	len = scsi_2btoul(cdb->length);
 	byte2 = cdb->byte2;
 
 	/*
 	 * If we've got a kernel request that hasn't been malloced yet,
 	 * malloc it and tell the caller the data buffer is here.
 	 */
 	if ((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0) {
 		ctsio->kern_data_ptr = malloc(len, M_CTL, M_WAITOK);;
 		ctsio->kern_data_len = len;
 		ctsio->kern_total_len = len;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	len = ctsio->kern_total_len - ctsio->kern_data_resid;
 	hdr = (struct scsi_unmap_header *)ctsio->kern_data_ptr;
 	if (len < sizeof (*hdr) ||
 	    len < (scsi_2btoul(hdr->length) + sizeof(hdr->length)) ||
 	    len < (scsi_2btoul(hdr->desc_length) + sizeof (*hdr)) ||
 	    scsi_2btoul(hdr->desc_length) % sizeof(*buf) != 0) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 0,
 				      /*command*/ 0,
 				      /*field*/ 0,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		goto done;
 	}
 	len = scsi_2btoul(hdr->desc_length);
 	buf = (struct scsi_unmap_desc *)(hdr + 1);
 	end = buf + len / sizeof(*buf);
 
 	endnz = buf;
 	for (range = buf; range < end; range++) {
 		lba = scsi_8btou64(range->lba);
 		num_blocks = scsi_4btoul(range->length);
 		if (((lba + num_blocks) > (lun->be_lun->maxlba + 1))
 		 || ((lba + num_blocks) < lba)) {
 			ctl_set_lba_out_of_range(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		if (num_blocks != 0)
 			endnz = range + 1;
 	}
 
 	/*
 	 * Block backend can not handle zero last range.
 	 * Filter it out and return if there is nothing left.
 	 */
 	len = (uint8_t *)endnz - (uint8_t *)buf;
 	if (len == 0) {
 		ctl_set_success(ctsio);
 		goto done;
 	}
 
 	mtx_lock(&lun->lun_lock);
 	ptrlen = (struct ctl_ptr_len_flags *)
 	    &ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	ptrlen->ptr = (void *)buf;
 	ptrlen->len = len;
 	ptrlen->flags = byte2;
 	ctl_check_blocked(lun);
 	mtx_unlock(&lun->lun_lock);
 
 	retval = lun->backend->config_write((union ctl_io *)ctsio);
 	return (retval);
 
 done:
 	if (ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) {
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctsio->io_hdr.flags &= ~CTL_FLAG_ALLOCATED;
 	}
 	ctl_done((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 /*
  * Note that this function currently doesn't actually do anything inside
  * CTL to enforce things if the DQue bit is turned on.
  *
  * Also note that this function can't be used in the default case, because
  * the DQue bit isn't set in the changeable mask for the control mode page
  * anyway.  This is just here as an example for how to implement a page
  * handler, and a placeholder in case we want to allow the user to turn
  * tagged queueing on and off.
  *
  * The D_SENSE bit handling is functional, however, and will turn
  * descriptor sense on and off for a given LUN.
  */
 int
 ctl_control_page_handler(struct ctl_scsiio *ctsio,
 			 struct ctl_page_index *page_index, uint8_t *page_ptr)
 {
 	struct scsi_control_page *current_cp, *saved_cp, *user_cp;
 	struct ctl_lun *lun;
 	int set_ua;
 	uint32_t initidx;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	initidx = ctl_get_initindex(&ctsio->io_hdr.nexus);
 	set_ua = 0;
 
 	user_cp = (struct scsi_control_page *)page_ptr;
 	current_cp = (struct scsi_control_page *)
 		(page_index->page_data + (page_index->page_len *
 		CTL_PAGE_CURRENT));
 	saved_cp = (struct scsi_control_page *)
 		(page_index->page_data + (page_index->page_len *
 		CTL_PAGE_SAVED));
 
 	mtx_lock(&lun->lun_lock);
 	if (((current_cp->rlec & SCP_DSENSE) == 0)
 	 && ((user_cp->rlec & SCP_DSENSE) != 0)) {
 		/*
 		 * Descriptor sense is currently turned off and the user
 		 * wants to turn it on.
 		 */
 		current_cp->rlec |= SCP_DSENSE;
 		saved_cp->rlec |= SCP_DSENSE;
 		lun->flags |= CTL_LUN_SENSE_DESC;
 		set_ua = 1;
 	} else if (((current_cp->rlec & SCP_DSENSE) != 0)
 		&& ((user_cp->rlec & SCP_DSENSE) == 0)) {
 		/*
 		 * Descriptor sense is currently turned on, and the user
 		 * wants to turn it off.
 		 */
 		current_cp->rlec &= ~SCP_DSENSE;
 		saved_cp->rlec &= ~SCP_DSENSE;
 		lun->flags &= ~CTL_LUN_SENSE_DESC;
 		set_ua = 1;
 	}
 	if ((current_cp->queue_flags & SCP_QUEUE_ALG_MASK) !=
 	    (user_cp->queue_flags & SCP_QUEUE_ALG_MASK)) {
 		current_cp->queue_flags &= ~SCP_QUEUE_ALG_MASK;
 		current_cp->queue_flags |= user_cp->queue_flags & SCP_QUEUE_ALG_MASK;
 		saved_cp->queue_flags &= ~SCP_QUEUE_ALG_MASK;
 		saved_cp->queue_flags |= user_cp->queue_flags & SCP_QUEUE_ALG_MASK;
 		set_ua = 1;
 	}
 	if ((current_cp->eca_and_aen & SCP_SWP) !=
 	    (user_cp->eca_and_aen & SCP_SWP)) {
 		current_cp->eca_and_aen &= ~SCP_SWP;
 		current_cp->eca_and_aen |= user_cp->eca_and_aen & SCP_SWP;
 		saved_cp->eca_and_aen &= ~SCP_SWP;
 		saved_cp->eca_and_aen |= user_cp->eca_and_aen & SCP_SWP;
 		set_ua = 1;
 	}
 	if (set_ua != 0)
 		ctl_est_ua_all(lun, initidx, CTL_UA_MODE_CHANGE);
 	mtx_unlock(&lun->lun_lock);
 
 	return (0);
 }
 
 int
 ctl_caching_sp_handler(struct ctl_scsiio *ctsio,
 		     struct ctl_page_index *page_index, uint8_t *page_ptr)
 {
 	struct scsi_caching_page *current_cp, *saved_cp, *user_cp;
 	struct ctl_lun *lun;
 	int set_ua;
 	uint32_t initidx;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	initidx = ctl_get_initindex(&ctsio->io_hdr.nexus);
 	set_ua = 0;
 
 	user_cp = (struct scsi_caching_page *)page_ptr;
 	current_cp = (struct scsi_caching_page *)
 		(page_index->page_data + (page_index->page_len *
 		CTL_PAGE_CURRENT));
 	saved_cp = (struct scsi_caching_page *)
 		(page_index->page_data + (page_index->page_len *
 		CTL_PAGE_SAVED));
 
 	mtx_lock(&lun->lun_lock);
 	if ((current_cp->flags1 & (SCP_WCE | SCP_RCD)) !=
 	    (user_cp->flags1 & (SCP_WCE | SCP_RCD))) {
 		current_cp->flags1 &= ~(SCP_WCE | SCP_RCD);
 		current_cp->flags1 |= user_cp->flags1 & (SCP_WCE | SCP_RCD);
 		saved_cp->flags1 &= ~(SCP_WCE | SCP_RCD);
 		saved_cp->flags1 |= user_cp->flags1 & (SCP_WCE | SCP_RCD);
 		set_ua = 1;
 	}
 	if (set_ua != 0)
 		ctl_est_ua_all(lun, initidx, CTL_UA_MODE_CHANGE);
 	mtx_unlock(&lun->lun_lock);
 
 	return (0);
 }
 
 int
 ctl_debugconf_sp_select_handler(struct ctl_scsiio *ctsio,
 				struct ctl_page_index *page_index,
 				uint8_t *page_ptr)
 {
 	uint8_t *c;
 	int i;
 
 	c = ((struct copan_debugconf_subpage *)page_ptr)->ctl_time_io_secs;
 	ctl_time_io_secs =
 		(c[0] << 8) |
 		(c[1] << 0) |
 		0;
 	CTL_DEBUG_PRINT(("set ctl_time_io_secs to %d\n", ctl_time_io_secs));
 	printf("set ctl_time_io_secs to %d\n", ctl_time_io_secs);
 	printf("page data:");
 	for (i=0; i<8; i++)
 		printf(" %.2x",page_ptr[i]);
 	printf("\n");
 	return (0);
 }
 
 int
 ctl_debugconf_sp_sense_handler(struct ctl_scsiio *ctsio,
 			       struct ctl_page_index *page_index,
 			       int pc)
 {
 	struct copan_debugconf_subpage *page;
 
 	page = (struct copan_debugconf_subpage *)page_index->page_data +
 		(page_index->page_len * pc);
 
 	switch (pc) {
 	case SMS_PAGE_CTRL_CHANGEABLE >> 6:
 	case SMS_PAGE_CTRL_DEFAULT >> 6:
 	case SMS_PAGE_CTRL_SAVED >> 6:
 		/*
 		 * We don't update the changable or default bits for this page.
 		 */
 		break;
 	case SMS_PAGE_CTRL_CURRENT >> 6:
 		page->ctl_time_io_secs[0] = ctl_time_io_secs >> 8;
 		page->ctl_time_io_secs[1] = ctl_time_io_secs >> 0;
 		break;
 	default:
 #ifdef NEEDTOPORT
 		EPRINT(0, "Invalid PC %d!!", pc);
 #endif /* NEEDTOPORT */
 		break;
 	}
 	return (0);
 }
 
 
 static int
 ctl_do_mode_select(union ctl_io *io)
 {
 	struct scsi_mode_page_header *page_header;
 	struct ctl_page_index *page_index;
 	struct ctl_scsiio *ctsio;
 	int control_dev, page_len;
 	int page_len_offset, page_len_size;
 	union ctl_modepage_info *modepage_info;
 	struct ctl_lun *lun;
 	int *len_left, *len_used;
 	int retval, i;
 
 	ctsio = &io->scsiio;
 	page_index = NULL;
 	page_len = 0;
 	retval = CTL_RETVAL_COMPLETE;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	if (lun->be_lun->lun_type != T_DIRECT)
 		control_dev = 1;
 	else
 		control_dev = 0;
 
 	modepage_info = (union ctl_modepage_info *)
 		ctsio->io_hdr.ctl_private[CTL_PRIV_MODEPAGE].bytes;
 	len_left = &modepage_info->header.len_left;
 	len_used = &modepage_info->header.len_used;
 
 do_next_page:
 
 	page_header = (struct scsi_mode_page_header *)
 		(ctsio->kern_data_ptr + *len_used);
 
 	if (*len_left == 0) {
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	} else if (*len_left < sizeof(struct scsi_mode_page_header)) {
 
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_set_param_len_error(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 
 	} else if ((page_header->page_code & SMPH_SPF)
 		&& (*len_left < sizeof(struct scsi_mode_page_header_sp))) {
 
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_set_param_len_error(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 
 	/*
 	 * XXX KDM should we do something with the block descriptor?
 	 */
 	for (i = 0; i < CTL_NUM_MODE_PAGES; i++) {
 
 		if ((control_dev != 0)
 		 && (lun->mode_pages.index[i].page_flags &
 		     CTL_PAGE_FLAG_DISK_ONLY))
 			continue;
 
 		if ((lun->mode_pages.index[i].page_code & SMPH_PC_MASK) !=
 		    (page_header->page_code & SMPH_PC_MASK))
 			continue;
 
 		/*
 		 * If neither page has a subpage code, then we've got a
 		 * match.
 		 */
 		if (((lun->mode_pages.index[i].page_code & SMPH_SPF) == 0)
 		 && ((page_header->page_code & SMPH_SPF) == 0)) {
 			page_index = &lun->mode_pages.index[i];
 			page_len = page_header->page_length;
 			break;
 		}
 
 		/*
 		 * If both pages have subpages, then the subpage numbers
 		 * have to match.
 		 */
 		if ((lun->mode_pages.index[i].page_code & SMPH_SPF)
 		  && (page_header->page_code & SMPH_SPF)) {
 			struct scsi_mode_page_header_sp *sph;
 
 			sph = (struct scsi_mode_page_header_sp *)page_header;
 
 			if (lun->mode_pages.index[i].subpage ==
 			    sph->subpage) {
 				page_index = &lun->mode_pages.index[i];
 				page_len = scsi_2btoul(sph->page_length);
 				break;
 			}
 		}
 	}
 
 	/*
 	 * If we couldn't find the page, or if we don't have a mode select
 	 * handler for it, send back an error to the user.
 	 */
 	if ((page_index == NULL)
 	 || (page_index->select_handler == NULL)) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 0,
 				      /*field*/ *len_used,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	if (page_index->page_code & SMPH_SPF) {
 		page_len_offset = 2;
 		page_len_size = 2;
 	} else {
 		page_len_size = 1;
 		page_len_offset = 1;
 	}
 
 	/*
 	 * If the length the initiator gives us isn't the one we specify in
 	 * the mode page header, or if they didn't specify enough data in
 	 * the CDB to avoid truncating this page, kick out the request.
 	 */
 	if ((page_len != (page_index->page_len - page_len_offset -
 			  page_len_size))
 	 || (*len_left < page_index->page_len)) {
 
 
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 0,
 				      /*field*/ *len_used + page_len_offset,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * Run through the mode page, checking to make sure that the bits
 	 * the user changed are actually legal for him to change.
 	 */
 	for (i = 0; i < page_index->page_len; i++) {
 		uint8_t *user_byte, *change_mask, *current_byte;
 		int bad_bit;
 		int j;
 
 		user_byte = (uint8_t *)page_header + i;
 		change_mask = page_index->page_data +
 			      (page_index->page_len * CTL_PAGE_CHANGEABLE) + i;
 		current_byte = page_index->page_data +
 			       (page_index->page_len * CTL_PAGE_CURRENT) + i;
 
 		/*
 		 * Check to see whether the user set any bits in this byte
 		 * that he is not allowed to set.
 		 */
 		if ((*user_byte & ~(*change_mask)) ==
 		    (*current_byte & ~(*change_mask)))
 			continue;
 
 		/*
 		 * Go through bit by bit to determine which one is illegal.
 		 */
 		bad_bit = 0;
 		for (j = 7; j >= 0; j--) {
 			if ((((1 << i) & ~(*change_mask)) & *user_byte) !=
 			    (((1 << i) & ~(*change_mask)) & *current_byte)) {
 				bad_bit = i;
 				break;
 			}
 		}
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 0,
 				      /*field*/ *len_used + i,
 				      /*bit_valid*/ 1,
 				      /*bit*/ bad_bit);
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * Decrement these before we call the page handler, since we may
 	 * end up getting called back one way or another before the handler
 	 * returns to this context.
 	 */
 	*len_left -= page_index->page_len;
 	*len_used += page_index->page_len;
 
 	retval = page_index->select_handler(ctsio, page_index,
 					    (uint8_t *)page_header);
 
 	/*
 	 * If the page handler returns CTL_RETVAL_QUEUED, then we need to
 	 * wait until this queued command completes to finish processing
 	 * the mode page.  If it returns anything other than
 	 * CTL_RETVAL_COMPLETE (e.g. CTL_RETVAL_ERROR), then it should have
 	 * already set the sense information, freed the data pointer, and
 	 * completed the io for us.
 	 */
 	if (retval != CTL_RETVAL_COMPLETE)
 		goto bailout_no_done;
 
 	/*
 	 * If the initiator sent us more than one page, parse the next one.
 	 */
 	if (*len_left > 0)
 		goto do_next_page;
 
 	ctl_set_success(ctsio);
 	free(ctsio->kern_data_ptr, M_CTL);
 	ctl_done((union ctl_io *)ctsio);
 
 bailout_no_done:
 
 	return (CTL_RETVAL_COMPLETE);
 
 }
 
 int
 ctl_mode_select(struct ctl_scsiio *ctsio)
 {
 	int param_len, pf, sp;
 	int header_size, bd_len;
 	int len_left, len_used;
 	struct ctl_page_index *page_index;
 	struct ctl_lun *lun;
 	int control_dev, page_len;
 	union ctl_modepage_info *modepage_info;
 	int retval;
 
 	pf = 0;
 	sp = 0;
 	page_len = 0;
 	len_used = 0;
 	len_left = 0;
 	retval = 0;
 	bd_len = 0;
 	page_index = NULL;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	if (lun->be_lun->lun_type != T_DIRECT)
 		control_dev = 1;
 	else
 		control_dev = 0;
 
 	switch (ctsio->cdb[0]) {
 	case MODE_SELECT_6: {
 		struct scsi_mode_select_6 *cdb;
 
 		cdb = (struct scsi_mode_select_6 *)ctsio->cdb;
 
 		pf = (cdb->byte2 & SMS_PF) ? 1 : 0;
 		sp = (cdb->byte2 & SMS_SP) ? 1 : 0;
 
 		param_len = cdb->length;
 		header_size = sizeof(struct scsi_mode_header_6);
 		break;
 	}
 	case MODE_SELECT_10: {
 		struct scsi_mode_select_10 *cdb;
 
 		cdb = (struct scsi_mode_select_10 *)ctsio->cdb;
 
 		pf = (cdb->byte2 & SMS_PF) ? 1 : 0;
 		sp = (cdb->byte2 & SMS_SP) ? 1 : 0;
 
 		param_len = scsi_2btoul(cdb->length);
 		header_size = sizeof(struct scsi_mode_header_10);
 		break;
 	}
 	default:
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 		break; /* NOTREACHED */
 	}
 
 	/*
 	 * From SPC-3:
 	 * "A parameter list length of zero indicates that the Data-Out Buffer
 	 * shall be empty. This condition shall not be considered as an error."
 	 */
 	if (param_len == 0) {
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * Since we'll hit this the first time through, prior to
 	 * allocation, we don't need to free a data buffer here.
 	 */
 	if (param_len < header_size) {
 		ctl_set_param_len_error(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * Allocate the data buffer and grab the user's data.  In theory,
 	 * we shouldn't have to sanity check the parameter list length here
 	 * because the maximum size is 64K.  We should be able to malloc
 	 * that much without too many problems.
 	 */
 	if ((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0) {
 		ctsio->kern_data_ptr = malloc(param_len, M_CTL, M_WAITOK);
 		ctsio->kern_data_len = param_len;
 		ctsio->kern_total_len = param_len;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	switch (ctsio->cdb[0]) {
 	case MODE_SELECT_6: {
 		struct scsi_mode_header_6 *mh6;
 
 		mh6 = (struct scsi_mode_header_6 *)ctsio->kern_data_ptr;
 		bd_len = mh6->blk_desc_len;
 		break;
 	}
 	case MODE_SELECT_10: {
 		struct scsi_mode_header_10 *mh10;
 
 		mh10 = (struct scsi_mode_header_10 *)ctsio->kern_data_ptr;
 		bd_len = scsi_2btoul(mh10->blk_desc_len);
 		break;
 	}
 	default:
 		panic("Invalid CDB type %#x", ctsio->cdb[0]);
 		break;
 	}
 
 	if (param_len < (header_size + bd_len)) {
 		free(ctsio->kern_data_ptr, M_CTL);
 		ctl_set_param_len_error(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * Set the IO_CONT flag, so that if this I/O gets passed to
 	 * ctl_config_write_done(), it'll get passed back to
 	 * ctl_do_mode_select() for further processing, or completion if
 	 * we're all done.
 	 */
 	ctsio->io_hdr.flags |= CTL_FLAG_IO_CONT;
 	ctsio->io_cont = ctl_do_mode_select;
 
 	modepage_info = (union ctl_modepage_info *)
 		ctsio->io_hdr.ctl_private[CTL_PRIV_MODEPAGE].bytes;
 
 	memset(modepage_info, 0, sizeof(*modepage_info));
 
 	len_left = param_len - header_size - bd_len;
 	len_used = header_size + bd_len;
 
 	modepage_info->header.len_left = len_left;
 	modepage_info->header.len_used = len_used;
 
 	return (ctl_do_mode_select((union ctl_io *)ctsio));
 }
 
 int
 ctl_mode_sense(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	int pc, page_code, dbd, llba, subpage;
 	int alloc_len, page_len, header_len, total_len;
 	struct scsi_mode_block_descr *block_desc;
 	struct ctl_page_index *page_index;
 	int control_dev;
 
 	dbd = 0;
 	llba = 0;
 	block_desc = NULL;
 	page_index = NULL;
 
 	CTL_DEBUG_PRINT(("ctl_mode_sense\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	if (lun->be_lun->lun_type != T_DIRECT)
 		control_dev = 1;
 	else
 		control_dev = 0;
 
 	switch (ctsio->cdb[0]) {
 	case MODE_SENSE_6: {
 		struct scsi_mode_sense_6 *cdb;
 
 		cdb = (struct scsi_mode_sense_6 *)ctsio->cdb;
 
 		header_len = sizeof(struct scsi_mode_hdr_6);
 		if (cdb->byte2 & SMS_DBD)
 			dbd = 1;
 		else
 			header_len += sizeof(struct scsi_mode_block_descr);
 
 		pc = (cdb->page & SMS_PAGE_CTRL_MASK) >> 6;
 		page_code = cdb->page & SMS_PAGE_CODE;
 		subpage = cdb->subpage;
 		alloc_len = cdb->length;
 		break;
 	}
 	case MODE_SENSE_10: {
 		struct scsi_mode_sense_10 *cdb;
 
 		cdb = (struct scsi_mode_sense_10 *)ctsio->cdb;
 
 		header_len = sizeof(struct scsi_mode_hdr_10);
 
 		if (cdb->byte2 & SMS_DBD)
 			dbd = 1;
 		else
 			header_len += sizeof(struct scsi_mode_block_descr);
 		if (cdb->byte2 & SMS10_LLBAA)
 			llba = 1;
 		pc = (cdb->page & SMS_PAGE_CTRL_MASK) >> 6;
 		page_code = cdb->page & SMS_PAGE_CODE;
 		subpage = cdb->subpage;
 		alloc_len = scsi_2btoul(cdb->length);
 		break;
 	}
 	default:
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 		break; /* NOTREACHED */
 	}
 
 	/*
 	 * We have to make a first pass through to calculate the size of
 	 * the pages that match the user's query.  Then we allocate enough
 	 * memory to hold it, and actually copy the data into the buffer.
 	 */
 	switch (page_code) {
 	case SMS_ALL_PAGES_PAGE: {
 		int i;
 
 		page_len = 0;
 
 		/*
 		 * At the moment, values other than 0 and 0xff here are
 		 * reserved according to SPC-3.
 		 */
 		if ((subpage != SMS_SUBPAGE_PAGE_0)
 		 && (subpage != SMS_SUBPAGE_ALL)) {
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 3,
 					      /*bit_valid*/ 0,
 					      /*bit*/ 0);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 
 		for (i = 0; i < CTL_NUM_MODE_PAGES; i++) {
 			if ((control_dev != 0)
 			 && (lun->mode_pages.index[i].page_flags &
 			     CTL_PAGE_FLAG_DISK_ONLY))
 				continue;
 
 			/*
 			 * We don't use this subpage if the user didn't
 			 * request all subpages.
 			 */
 			if ((lun->mode_pages.index[i].subpage != 0)
 			 && (subpage == SMS_SUBPAGE_PAGE_0))
 				continue;
 
 #if 0
 			printf("found page %#x len %d\n",
 			       lun->mode_pages.index[i].page_code &
 			       SMPH_PC_MASK,
 			       lun->mode_pages.index[i].page_len);
 #endif
 			page_len += lun->mode_pages.index[i].page_len;
 		}
 		break;
 	}
 	default: {
 		int i;
 
 		page_len = 0;
 
 		for (i = 0; i < CTL_NUM_MODE_PAGES; i++) {
 			/* Look for the right page code */
 			if ((lun->mode_pages.index[i].page_code &
 			     SMPH_PC_MASK) != page_code)
 				continue;
 
 			/* Look for the right subpage or the subpage wildcard*/
 			if ((lun->mode_pages.index[i].subpage != subpage)
 			 && (subpage != SMS_SUBPAGE_ALL))
 				continue;
 
 			/* Make sure the page is supported for this dev type */
 			if ((control_dev != 0)
 			 && (lun->mode_pages.index[i].page_flags &
 			     CTL_PAGE_FLAG_DISK_ONLY))
 				continue;
 
 #if 0
 			printf("found page %#x len %d\n",
 			       lun->mode_pages.index[i].page_code &
 			       SMPH_PC_MASK,
 			       lun->mode_pages.index[i].page_len);
 #endif
 
 			page_len += lun->mode_pages.index[i].page_len;
 		}
 
 		if (page_len == 0) {
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 2,
 					      /*bit_valid*/ 1,
 					      /*bit*/ 5);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		break;
 	}
 	}
 
 	total_len = header_len + page_len;
 #if 0
 	printf("header_len = %d, page_len = %d, total_len = %d\n",
 	       header_len, page_len, total_len);
 #endif
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 	ctsio->kern_sg_entries = 0;
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 
 	switch (ctsio->cdb[0]) {
 	case MODE_SENSE_6: {
 		struct scsi_mode_hdr_6 *header;
 
 		header = (struct scsi_mode_hdr_6 *)ctsio->kern_data_ptr;
 
 		header->datalen = MIN(total_len - 1, 254);
 		if (control_dev == 0) {
 			header->dev_specific = 0x10; /* DPOFUA */
 			if ((lun->flags & CTL_LUN_READONLY) ||
 			    (lun->mode_pages.control_page[CTL_PAGE_CURRENT]
 			    .eca_and_aen & SCP_SWP) != 0)
 				    header->dev_specific |= 0x80; /* WP */
 		}
 		if (dbd)
 			header->block_descr_len = 0;
 		else
 			header->block_descr_len =
 				sizeof(struct scsi_mode_block_descr);
 		block_desc = (struct scsi_mode_block_descr *)&header[1];
 		break;
 	}
 	case MODE_SENSE_10: {
 		struct scsi_mode_hdr_10 *header;
 		int datalen;
 
 		header = (struct scsi_mode_hdr_10 *)ctsio->kern_data_ptr;
 
 		datalen = MIN(total_len - 2, 65533);
 		scsi_ulto2b(datalen, header->datalen);
 		if (control_dev == 0) {
 			header->dev_specific = 0x10; /* DPOFUA */
 			if ((lun->flags & CTL_LUN_READONLY) ||
 			    (lun->mode_pages.control_page[CTL_PAGE_CURRENT]
 			    .eca_and_aen & SCP_SWP) != 0)
 				    header->dev_specific |= 0x80; /* WP */
 		}
 		if (dbd)
 			scsi_ulto2b(0, header->block_descr_len);
 		else
 			scsi_ulto2b(sizeof(struct scsi_mode_block_descr),
 				    header->block_descr_len);
 		block_desc = (struct scsi_mode_block_descr *)&header[1];
 		break;
 	}
 	default:
 		panic("invalid CDB type %#x", ctsio->cdb[0]);
 		break; /* NOTREACHED */
 	}
 
 	/*
 	 * If we've got a disk, use its blocksize in the block
 	 * descriptor.  Otherwise, just set it to 0.
 	 */
 	if (dbd == 0) {
 		if (control_dev == 0)
 			scsi_ulto3b(lun->be_lun->blocksize,
 				    block_desc->block_len);
 		else
 			scsi_ulto3b(0, block_desc->block_len);
 	}
 
 	switch (page_code) {
 	case SMS_ALL_PAGES_PAGE: {
 		int i, data_used;
 
 		data_used = header_len;
 		for (i = 0; i < CTL_NUM_MODE_PAGES; i++) {
 			struct ctl_page_index *page_index;
 
 			page_index = &lun->mode_pages.index[i];
 
 			if ((control_dev != 0)
 			 && (page_index->page_flags &
 			    CTL_PAGE_FLAG_DISK_ONLY))
 				continue;
 
 			/*
 			 * We don't use this subpage if the user didn't
 			 * request all subpages.  We already checked (above)
 			 * to make sure the user only specified a subpage
 			 * of 0 or 0xff in the SMS_ALL_PAGES_PAGE case.
 			 */
 			if ((page_index->subpage != 0)
 			 && (subpage == SMS_SUBPAGE_PAGE_0))
 				continue;
 
 			/*
 			 * Call the handler, if it exists, to update the
 			 * page to the latest values.
 			 */
 			if (page_index->sense_handler != NULL)
 				page_index->sense_handler(ctsio, page_index,pc);
 
 			memcpy(ctsio->kern_data_ptr + data_used,
 			       page_index->page_data +
 			       (page_index->page_len * pc),
 			       page_index->page_len);
 			data_used += page_index->page_len;
 		}
 		break;
 	}
 	default: {
 		int i, data_used;
 
 		data_used = header_len;
 
 		for (i = 0; i < CTL_NUM_MODE_PAGES; i++) {
 			struct ctl_page_index *page_index;
 
 			page_index = &lun->mode_pages.index[i];
 
 			/* Look for the right page code */
 			if ((page_index->page_code & SMPH_PC_MASK) != page_code)
 				continue;
 
 			/* Look for the right subpage or the subpage wildcard*/
 			if ((page_index->subpage != subpage)
 			 && (subpage != SMS_SUBPAGE_ALL))
 				continue;
 
 			/* Make sure the page is supported for this dev type */
 			if ((control_dev != 0)
 			 && (page_index->page_flags &
 			     CTL_PAGE_FLAG_DISK_ONLY))
 				continue;
 
 			/*
 			 * Call the handler, if it exists, to update the
 			 * page to the latest values.
 			 */
 			if (page_index->sense_handler != NULL)
 				page_index->sense_handler(ctsio, page_index,pc);
 
 			memcpy(ctsio->kern_data_ptr + data_used,
 			       page_index->page_data +
 			       (page_index->page_len * pc),
 			       page_index->page_len);
 			data_used += page_index->page_len;
 		}
 		break;
 	}
 	}
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_lbp_log_sense_handler(struct ctl_scsiio *ctsio,
 			       struct ctl_page_index *page_index,
 			       int pc)
 {
 	struct ctl_lun *lun;
 	struct scsi_log_param_header *phdr;
 	uint8_t *data;
 	uint64_t val;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	data = page_index->page_data;
 
 	if (lun->backend->lun_attr != NULL &&
 	    (val = lun->backend->lun_attr(lun->be_lun->be_lun, "blocksavail"))
 	     != UINT64_MAX) {
 		phdr = (struct scsi_log_param_header *)data;
 		scsi_ulto2b(0x0001, phdr->param_code);
 		phdr->param_control = SLP_LBIN | SLP_LP;
 		phdr->param_len = 8;
 		data = (uint8_t *)(phdr + 1);
 		scsi_ulto4b(val >> CTL_LBP_EXPONENT, data);
 		data[4] = 0x02; /* per-pool */
 		data += phdr->param_len;
 	}
 
 	if (lun->backend->lun_attr != NULL &&
 	    (val = lun->backend->lun_attr(lun->be_lun->be_lun, "blocksused"))
 	     != UINT64_MAX) {
 		phdr = (struct scsi_log_param_header *)data;
 		scsi_ulto2b(0x0002, phdr->param_code);
 		phdr->param_control = SLP_LBIN | SLP_LP;
 		phdr->param_len = 8;
 		data = (uint8_t *)(phdr + 1);
 		scsi_ulto4b(val >> CTL_LBP_EXPONENT, data);
 		data[4] = 0x01; /* per-LUN */
 		data += phdr->param_len;
 	}
 
 	if (lun->backend->lun_attr != NULL &&
 	    (val = lun->backend->lun_attr(lun->be_lun->be_lun, "poolblocksavail"))
 	     != UINT64_MAX) {
 		phdr = (struct scsi_log_param_header *)data;
 		scsi_ulto2b(0x00f1, phdr->param_code);
 		phdr->param_control = SLP_LBIN | SLP_LP;
 		phdr->param_len = 8;
 		data = (uint8_t *)(phdr + 1);
 		scsi_ulto4b(val >> CTL_LBP_EXPONENT, data);
 		data[4] = 0x02; /* per-pool */
 		data += phdr->param_len;
 	}
 
 	if (lun->backend->lun_attr != NULL &&
 	    (val = lun->backend->lun_attr(lun->be_lun->be_lun, "poolblocksused"))
 	     != UINT64_MAX) {
 		phdr = (struct scsi_log_param_header *)data;
 		scsi_ulto2b(0x00f2, phdr->param_code);
 		phdr->param_control = SLP_LBIN | SLP_LP;
 		phdr->param_len = 8;
 		data = (uint8_t *)(phdr + 1);
 		scsi_ulto4b(val >> CTL_LBP_EXPONENT, data);
 		data[4] = 0x02; /* per-pool */
 		data += phdr->param_len;
 	}
 
 	page_index->page_len = data - page_index->page_data;
 	return (0);
 }
 
 int
 ctl_log_sense(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	int i, pc, page_code, subpage;
 	int alloc_len, total_len;
 	struct ctl_page_index *page_index;
 	struct scsi_log_sense *cdb;
 	struct scsi_log_header *header;
 
 	CTL_DEBUG_PRINT(("ctl_log_sense\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_log_sense *)ctsio->cdb;
 	pc = (cdb->page & SLS_PAGE_CTRL_MASK) >> 6;
 	page_code = cdb->page & SLS_PAGE_CODE;
 	subpage = cdb->subpage;
 	alloc_len = scsi_2btoul(cdb->length);
 
 	page_index = NULL;
 	for (i = 0; i < CTL_NUM_LOG_PAGES; i++) {
 		page_index = &lun->log_pages.index[i];
 
 		/* Look for the right page code */
 		if ((page_index->page_code & SL_PAGE_CODE) != page_code)
 			continue;
 
 		/* Look for the right subpage or the subpage wildcard*/
 		if (page_index->subpage != subpage)
 			continue;
 
 		break;
 	}
 	if (i >= CTL_NUM_LOG_PAGES) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	total_len = sizeof(struct scsi_log_header) + page_index->page_len;
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 	ctsio->kern_sg_entries = 0;
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 
 	header = (struct scsi_log_header *)ctsio->kern_data_ptr;
 	header->page = page_index->page_code;
 	if (page_index->subpage) {
 		header->page |= SL_SPF;
 		header->subpage = page_index->subpage;
 	}
 	scsi_ulto2b(page_index->page_len, header->datalen);
 
 	/*
 	 * Call the handler, if it exists, to update the
 	 * page to the latest values.
 	 */
 	if (page_index->sense_handler != NULL)
 		page_index->sense_handler(ctsio, page_index, pc);
 
 	memcpy(header + 1, page_index->page_data, page_index->page_len);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_read_capacity(struct ctl_scsiio *ctsio)
 {
 	struct scsi_read_capacity *cdb;
 	struct scsi_read_capacity_data *data;
 	struct ctl_lun *lun;
 	uint32_t lba;
 
 	CTL_DEBUG_PRINT(("ctl_read_capacity\n"));
 
 	cdb = (struct scsi_read_capacity *)ctsio->cdb;
 
 	lba = scsi_4btoul(cdb->addr);
 	if (((cdb->pmi & SRC_PMI) == 0)
 	 && (lba != 0)) {
 		ctl_set_invalid_field(/*ctsio*/ ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	ctsio->kern_data_ptr = malloc(sizeof(*data), M_CTL, M_WAITOK | M_ZERO);
 	data = (struct scsi_read_capacity_data *)ctsio->kern_data_ptr;
 	ctsio->residual = 0;
 	ctsio->kern_data_len = sizeof(*data);
 	ctsio->kern_total_len = sizeof(*data);
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * If the maximum LBA is greater than 0xfffffffe, the user must
 	 * issue a SERVICE ACTION IN (16) command, with the read capacity
 	 * serivce action set.
 	 */
 	if (lun->be_lun->maxlba > 0xfffffffe)
 		scsi_ulto4b(0xffffffff, data->addr);
 	else
 		scsi_ulto4b(lun->be_lun->maxlba, data->addr);
 
 	/*
 	 * XXX KDM this may not be 512 bytes...
 	 */
 	scsi_ulto4b(lun->be_lun->blocksize, data->length);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_read_capacity_16(struct ctl_scsiio *ctsio)
 {
 	struct scsi_read_capacity_16 *cdb;
 	struct scsi_read_capacity_data_long *data;
 	struct ctl_lun *lun;
 	uint64_t lba;
 	uint32_t alloc_len;
 
 	CTL_DEBUG_PRINT(("ctl_read_capacity_16\n"));
 
 	cdb = (struct scsi_read_capacity_16 *)ctsio->cdb;
 
 	alloc_len = scsi_4btoul(cdb->alloc_len);
 	lba = scsi_8btou64(cdb->addr);
 
 	if ((cdb->reladr & SRC16_PMI)
 	 && (lba != 0)) {
 		ctl_set_invalid_field(/*ctsio*/ ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	ctsio->kern_data_ptr = malloc(sizeof(*data), M_CTL, M_WAITOK | M_ZERO);
 	data = (struct scsi_read_capacity_data_long *)ctsio->kern_data_ptr;
 
 	if (sizeof(*data) < alloc_len) {
 		ctsio->residual = alloc_len - sizeof(*data);
 		ctsio->kern_data_len = sizeof(*data);
 		ctsio->kern_total_len = sizeof(*data);
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	scsi_u64to8b(lun->be_lun->maxlba, data->addr);
 	/* XXX KDM this may not be 512 bytes... */
 	scsi_ulto4b(lun->be_lun->blocksize, data->length);
 	data->prot_lbppbe = lun->be_lun->pblockexp & SRC16_LBPPBE;
 	scsi_ulto2b(lun->be_lun->pblockoff & SRC16_LALBA_A, data->lalba_lbp);
 	if (lun->be_lun->flags & CTL_LUN_FLAG_UNMAP)
 		data->lalba_lbp[0] |= SRC16_LBPME | SRC16_LBPRZ;
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_get_lba_status(struct ctl_scsiio *ctsio)
 {
 	struct scsi_get_lba_status *cdb;
 	struct scsi_get_lba_status_data *data;
 	struct ctl_lun *lun;
 	struct ctl_lba_len_flags *lbalen;
 	uint64_t lba;
 	uint32_t alloc_len, total_len;
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_get_lba_status\n"));
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_get_lba_status *)ctsio->cdb;
 	lba = scsi_8btou64(cdb->addr);
 	alloc_len = scsi_4btoul(cdb->alloc_len);
 
 	if (lba > lun->be_lun->maxlba) {
 		ctl_set_lba_out_of_range(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	total_len = sizeof(*data) + sizeof(data->descr[0]);
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 	data = (struct scsi_get_lba_status_data *)ctsio->kern_data_ptr;
 
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/* Fill dummy data in case backend can't tell anything. */
 	scsi_ulto4b(4 + sizeof(data->descr[0]), data->length);
 	scsi_u64to8b(lba, data->descr[0].addr);
 	scsi_ulto4b(MIN(UINT32_MAX, lun->be_lun->maxlba + 1 - lba),
 	    data->descr[0].length);
 	data->descr[0].status = 0; /* Mapped or unknown. */
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 
 	lbalen = (struct ctl_lba_len_flags *)&ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	lbalen->lba = lba;
 	lbalen->len = total_len;
 	lbalen->flags = 0;
 	retval = lun->backend->config_read((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_read_defect(struct ctl_scsiio *ctsio)
 {
 	struct scsi_read_defect_data_10 *ccb10;
 	struct scsi_read_defect_data_12 *ccb12;
 	struct scsi_read_defect_data_hdr_10 *data10;
 	struct scsi_read_defect_data_hdr_12 *data12;
 	uint32_t alloc_len, data_len;
 	uint8_t format;
 
 	CTL_DEBUG_PRINT(("ctl_read_defect\n"));
 
 	if (ctsio->cdb[0] == READ_DEFECT_DATA_10) {
 		ccb10 = (struct scsi_read_defect_data_10 *)&ctsio->cdb;
 		format = ccb10->format;
 		alloc_len = scsi_2btoul(ccb10->alloc_length);
 		data_len = sizeof(*data10);
 	} else {
 		ccb12 = (struct scsi_read_defect_data_12 *)&ctsio->cdb;
 		format = ccb12->format;
 		alloc_len = scsi_4btoul(ccb12->alloc_length);
 		data_len = sizeof(*data12);
 	}
 	if (alloc_len == 0) {
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	if (ctsio->cdb[0] == READ_DEFECT_DATA_10) {
 		data10 = (struct scsi_read_defect_data_hdr_10 *)
 		    ctsio->kern_data_ptr;
 		data10->format = format;
 		scsi_ulto2b(0, data10->length);
 	} else {
 		data12 = (struct scsi_read_defect_data_hdr_12 *)
 		    ctsio->kern_data_ptr;
 		data12->format = format;
 		scsi_ulto2b(0, data12->generation);
 		scsi_ulto4b(0, data12->length);
 	}
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_report_tagret_port_groups(struct ctl_scsiio *ctsio)
 {
 	struct scsi_maintenance_in *cdb;
 	int retval;
 	int alloc_len, ext, total_len = 0, g, p, pc, pg, gs, os;
 	int num_target_port_groups, num_target_ports;
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	struct ctl_port *port;
 	struct scsi_target_group_data *rtg_ptr;
 	struct scsi_target_group_data_extended *rtg_ext_ptr;
 	struct scsi_target_port_group_descriptor *tpg_desc;
 
 	CTL_DEBUG_PRINT(("ctl_report_tagret_port_groups\n"));
 
 	cdb = (struct scsi_maintenance_in *)ctsio->cdb;
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	softc = lun->ctl_softc;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	switch (cdb->byte2 & STG_PDF_MASK) {
 	case STG_PDF_LENGTH:
 		ext = 0;
 		break;
 	case STG_PDF_EXTENDED:
 		ext = 1;
 		break;
 	default:
 		ctl_set_invalid_field(/*ctsio*/ ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 1,
 				      /*bit*/ 5);
 		ctl_done((union ctl_io *)ctsio);
 		return(retval);
 	}
 
 	if (softc->is_single)
 		num_target_port_groups = 1;
 	else
 		num_target_port_groups = NUM_TARGET_PORT_GROUPS;
 	num_target_ports = 0;
 	mtx_lock(&softc->ctl_lock);
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		if ((port->status & CTL_PORT_STATUS_ONLINE) == 0)
 			continue;
 		if (ctl_map_lun_back(softc, port->targ_port, lun->lun) >=
 		    CTL_MAX_LUNS)
 			continue;
 		num_target_ports++;
 	}
 	mtx_unlock(&softc->ctl_lock);
 
 	if (ext)
 		total_len = sizeof(struct scsi_target_group_data_extended);
 	else
 		total_len = sizeof(struct scsi_target_group_data);
 	total_len += sizeof(struct scsi_target_port_group_descriptor) *
 		num_target_port_groups +
 	    sizeof(struct scsi_target_port_descriptor) *
 		num_target_ports * num_target_port_groups;
 
 	alloc_len = scsi_4btoul(cdb->length);
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 
 	ctsio->kern_sg_entries = 0;
 
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 
 	if (ext) {
 		rtg_ext_ptr = (struct scsi_target_group_data_extended *)
 		    ctsio->kern_data_ptr;
 		scsi_ulto4b(total_len - 4, rtg_ext_ptr->length);
 		rtg_ext_ptr->format_type = 0x10;
 		rtg_ext_ptr->implicit_transition_time = 0;
 		tpg_desc = &rtg_ext_ptr->groups[0];
 	} else {
 		rtg_ptr = (struct scsi_target_group_data *)
 		    ctsio->kern_data_ptr;
 		scsi_ulto4b(total_len - 4, rtg_ptr->length);
 		tpg_desc = &rtg_ptr->groups[0];
 	}
 
 	mtx_lock(&softc->ctl_lock);
 	pg = softc->port_offset / CTL_MAX_PORTS;
 	if (softc->flags & CTL_FLAG_ACTIVE_SHELF) {
 		if (softc->ha_mode == CTL_HA_MODE_ACT_STBY) {
 			gs = TPG_ASYMMETRIC_ACCESS_OPTIMIZED;
 			os = TPG_ASYMMETRIC_ACCESS_STANDBY;
 		} else if (lun->flags & CTL_LUN_PRIMARY_SC) {
 			gs = TPG_ASYMMETRIC_ACCESS_OPTIMIZED;
 			os = TPG_ASYMMETRIC_ACCESS_NONOPTIMIZED;
 		} else {
 			gs = TPG_ASYMMETRIC_ACCESS_NONOPTIMIZED;
 			os = TPG_ASYMMETRIC_ACCESS_OPTIMIZED;
 		}
 	} else {
 		gs = TPG_ASYMMETRIC_ACCESS_STANDBY;
 		os = TPG_ASYMMETRIC_ACCESS_OPTIMIZED;
 	}
 	for (g = 0; g < num_target_port_groups; g++) {
 		tpg_desc->pref_state = (g == pg) ? gs : os;
 		tpg_desc->support = TPG_AO_SUP | TPG_AN_SUP | TPG_S_SUP;
 		scsi_ulto2b(g + 1, tpg_desc->target_port_group);
 		tpg_desc->status = TPG_IMPLICIT;
 		pc = 0;
 		STAILQ_FOREACH(port, &softc->port_list, links) {
 			if ((port->status & CTL_PORT_STATUS_ONLINE) == 0)
 				continue;
 			if (ctl_map_lun_back(softc, port->targ_port, lun->lun)
 			    >= CTL_MAX_LUNS)
 				continue;
 			p = port->targ_port % CTL_MAX_PORTS + g * CTL_MAX_PORTS;
 			scsi_ulto2b(p, tpg_desc->descriptors[pc].
 			    relative_target_port_identifier);
 			pc++;
 		}
 		tpg_desc->target_port_count = pc;
 		tpg_desc = (struct scsi_target_port_group_descriptor *)
 		    &tpg_desc->descriptors[pc];
 	}
 	mtx_unlock(&softc->ctl_lock);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return(retval);
 }
 
 int
 ctl_report_supported_opcodes(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct scsi_report_supported_opcodes *cdb;
 	const struct ctl_cmd_entry *entry, *sentry;
 	struct scsi_report_supported_opcodes_all *all;
 	struct scsi_report_supported_opcodes_descr *descr;
 	struct scsi_report_supported_opcodes_one *one;
 	int retval;
 	int alloc_len, total_len;
 	int opcode, service_action, i, j, num;
 
 	CTL_DEBUG_PRINT(("ctl_report_supported_opcodes\n"));
 
 	cdb = (struct scsi_report_supported_opcodes *)ctsio->cdb;
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	opcode = cdb->requested_opcode;
 	service_action = scsi_2btoul(cdb->requested_service_action);
 	switch (cdb->options & RSO_OPTIONS_MASK) {
 	case RSO_OPTIONS_ALL:
 		num = 0;
 		for (i = 0; i < 256; i++) {
 			entry = &ctl_cmd_table[i];
 			if (entry->flags & CTL_CMD_FLAG_SA5) {
 				for (j = 0; j < 32; j++) {
 					sentry = &((const struct ctl_cmd_entry *)
 					    entry->execute)[j];
 					if (ctl_cmd_applicable(
 					    lun->be_lun->lun_type, sentry))
 						num++;
 				}
 			} else {
 				if (ctl_cmd_applicable(lun->be_lun->lun_type,
 				    entry))
 					num++;
 			}
 		}
 		total_len = sizeof(struct scsi_report_supported_opcodes_all) +
 		    num * sizeof(struct scsi_report_supported_opcodes_descr);
 		break;
 	case RSO_OPTIONS_OC:
 		if (ctl_cmd_table[opcode].flags & CTL_CMD_FLAG_SA5) {
 			ctl_set_invalid_field(/*ctsio*/ ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 2,
 					      /*bit_valid*/ 1,
 					      /*bit*/ 2);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		total_len = sizeof(struct scsi_report_supported_opcodes_one) + 32;
 		break;
 	case RSO_OPTIONS_OC_SA:
 		if ((ctl_cmd_table[opcode].flags & CTL_CMD_FLAG_SA5) == 0 ||
 		    service_action >= 32) {
 			ctl_set_invalid_field(/*ctsio*/ ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 2,
 					      /*bit_valid*/ 1,
 					      /*bit*/ 2);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		total_len = sizeof(struct scsi_report_supported_opcodes_one) + 32;
 		break;
 	default:
 		ctl_set_invalid_field(/*ctsio*/ ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 1,
 				      /*bit*/ 2);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	alloc_len = scsi_4btoul(cdb->length);
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 
 	ctsio->kern_sg_entries = 0;
 
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 
 	switch (cdb->options & RSO_OPTIONS_MASK) {
 	case RSO_OPTIONS_ALL:
 		all = (struct scsi_report_supported_opcodes_all *)
 		    ctsio->kern_data_ptr;
 		num = 0;
 		for (i = 0; i < 256; i++) {
 			entry = &ctl_cmd_table[i];
 			if (entry->flags & CTL_CMD_FLAG_SA5) {
 				for (j = 0; j < 32; j++) {
 					sentry = &((const struct ctl_cmd_entry *)
 					    entry->execute)[j];
 					if (!ctl_cmd_applicable(
 					    lun->be_lun->lun_type, sentry))
 						continue;
 					descr = &all->descr[num++];
 					descr->opcode = i;
 					scsi_ulto2b(j, descr->service_action);
 					descr->flags = RSO_SERVACTV;
 					scsi_ulto2b(sentry->length,
 					    descr->cdb_length);
 				}
 			} else {
 				if (!ctl_cmd_applicable(lun->be_lun->lun_type,
 				    entry))
 					continue;
 				descr = &all->descr[num++];
 				descr->opcode = i;
 				scsi_ulto2b(0, descr->service_action);
 				descr->flags = 0;
 				scsi_ulto2b(entry->length, descr->cdb_length);
 			}
 		}
 		scsi_ulto4b(
 		    num * sizeof(struct scsi_report_supported_opcodes_descr),
 		    all->length);
 		break;
 	case RSO_OPTIONS_OC:
 		one = (struct scsi_report_supported_opcodes_one *)
 		    ctsio->kern_data_ptr;
 		entry = &ctl_cmd_table[opcode];
 		goto fill_one;
 	case RSO_OPTIONS_OC_SA:
 		one = (struct scsi_report_supported_opcodes_one *)
 		    ctsio->kern_data_ptr;
 		entry = &ctl_cmd_table[opcode];
 		entry = &((const struct ctl_cmd_entry *)
 		    entry->execute)[service_action];
 fill_one:
 		if (ctl_cmd_applicable(lun->be_lun->lun_type, entry)) {
 			one->support = 3;
 			scsi_ulto2b(entry->length, one->cdb_length);
 			one->cdb_usage[0] = opcode;
 			memcpy(&one->cdb_usage[1], entry->usage,
 			    entry->length - 1);
 		} else
 			one->support = 1;
 		break;
 	}
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return(retval);
 }
 
 int
 ctl_report_supported_tmf(struct ctl_scsiio *ctsio)
 {
 	struct scsi_report_supported_tmf *cdb;
 	struct scsi_report_supported_tmf_data *data;
 	int retval;
 	int alloc_len, total_len;
 
 	CTL_DEBUG_PRINT(("ctl_report_supported_tmf\n"));
 
 	cdb = (struct scsi_report_supported_tmf *)ctsio->cdb;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	total_len = sizeof(struct scsi_report_supported_tmf_data);
 	alloc_len = scsi_4btoul(cdb->length);
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 
 	ctsio->kern_sg_entries = 0;
 
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 
 	data = (struct scsi_report_supported_tmf_data *)ctsio->kern_data_ptr;
 	data->byte1 |= RST_ATS | RST_ATSS | RST_CTSS | RST_LURS | RST_TRS;
 	data->byte2 |= RST_ITNRS;
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (retval);
 }
 
 int
 ctl_report_timestamp(struct ctl_scsiio *ctsio)
 {
 	struct scsi_report_timestamp *cdb;
 	struct scsi_report_timestamp_data *data;
 	struct timeval tv;
 	int64_t timestamp;
 	int retval;
 	int alloc_len, total_len;
 
 	CTL_DEBUG_PRINT(("ctl_report_timestamp\n"));
 
 	cdb = (struct scsi_report_timestamp *)ctsio->cdb;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	total_len = sizeof(struct scsi_report_timestamp_data);
 	alloc_len = scsi_4btoul(cdb->length);
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 
 	ctsio->kern_sg_entries = 0;
 
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 
 	data = (struct scsi_report_timestamp_data *)ctsio->kern_data_ptr;
 	scsi_ulto2b(sizeof(*data) - 2, data->length);
 	data->origin = RTS_ORIG_OUTSIDE;
 	getmicrotime(&tv);
 	timestamp = (int64_t)tv.tv_sec * 1000 + tv.tv_usec / 1000;
 	scsi_ulto4b(timestamp >> 16, data->timestamp);
 	scsi_ulto2b(timestamp & 0xffff, &data->timestamp[4]);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (retval);
 }
 
 int
 ctl_persistent_reserve_in(struct ctl_scsiio *ctsio)
 {
 	struct scsi_per_res_in *cdb;
 	int alloc_len, total_len = 0;
 	/* struct scsi_per_res_in_rsrv in_data; */
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	uint64_t key;
 
 	CTL_DEBUG_PRINT(("ctl_persistent_reserve_in\n"));
 
 	cdb = (struct scsi_per_res_in *)ctsio->cdb;
 
 	alloc_len = scsi_2btoul(cdb->length);
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	softc = lun->ctl_softc;
 
 retry:
 	mtx_lock(&lun->lun_lock);
 	switch (cdb->action) {
 	case SPRI_RK: /* read keys */
 		total_len = sizeof(struct scsi_per_res_in_keys) +
 			lun->pr_key_count *
 			sizeof(struct scsi_per_res_key);
 		break;
 	case SPRI_RR: /* read reservation */
 		if (lun->flags & CTL_LUN_PR_RESERVED)
 			total_len = sizeof(struct scsi_per_res_in_rsrv);
 		else
 			total_len = sizeof(struct scsi_per_res_in_header);
 		break;
 	case SPRI_RC: /* report capabilities */
 		total_len = sizeof(struct scsi_per_res_cap);
 		break;
 	case SPRI_RS: /* read full status */
 		total_len = sizeof(struct scsi_per_res_in_header) +
 		    (sizeof(struct scsi_per_res_in_full_desc) + 256) *
 		    lun->pr_key_count;
 		break;
 	default:
 		panic("Invalid PR type %x", cdb->action);
 	}
 	mtx_unlock(&lun->lun_lock);
 
 	ctsio->kern_data_ptr = malloc(total_len, M_CTL, M_WAITOK | M_ZERO);
 
 	if (total_len < alloc_len) {
 		ctsio->residual = alloc_len - total_len;
 		ctsio->kern_data_len = total_len;
 		ctsio->kern_total_len = total_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	mtx_lock(&lun->lun_lock);
 	switch (cdb->action) {
 	case SPRI_RK: { // read keys
         struct scsi_per_res_in_keys *res_keys;
 		int i, key_count;
 
 		res_keys = (struct scsi_per_res_in_keys*)ctsio->kern_data_ptr;
 
 		/*
 		 * We had to drop the lock to allocate our buffer, which
 		 * leaves time for someone to come in with another
 		 * persistent reservation.  (That is unlikely, though,
 		 * since this should be the only persistent reservation
 		 * command active right now.)
 		 */
 		if (total_len != (sizeof(struct scsi_per_res_in_keys) +
 		    (lun->pr_key_count *
 		     sizeof(struct scsi_per_res_key)))){
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			printf("%s: reservation length changed, retrying\n",
 			       __func__);
 			goto retry;
 		}
 
 		scsi_ulto4b(lun->PRGeneration, res_keys->header.generation);
 
 		scsi_ulto4b(sizeof(struct scsi_per_res_key) *
 			     lun->pr_key_count, res_keys->header.length);
 
 		for (i = 0, key_count = 0; i < 2*CTL_MAX_INITIATORS; i++) {
 			if ((key = ctl_get_prkey(lun, i)) == 0)
 				continue;
 
 			/*
 			 * We used lun->pr_key_count to calculate the
 			 * size to allocate.  If it turns out the number of
 			 * initiators with the registered flag set is
 			 * larger than that (i.e. they haven't been kept in
 			 * sync), we've got a problem.
 			 */
 			if (key_count >= lun->pr_key_count) {
 #ifdef NEEDTOPORT
 				csevent_log(CSC_CTL | CSC_SHELF_SW |
 					    CTL_PR_ERROR,
 					    csevent_LogType_Fault,
 					    csevent_AlertLevel_Yellow,
 					    csevent_FRU_ShelfController,
 					    csevent_FRU_Firmware,
 				        csevent_FRU_Unknown,
 					    "registered keys %d >= key "
 					    "count %d", key_count,
 					    lun->pr_key_count);
 #endif
 				key_count++;
 				continue;
 			}
 			scsi_u64to8b(key, res_keys->keys[key_count].key);
 			key_count++;
 		}
 		break;
 	}
 	case SPRI_RR: { // read reservation
 		struct scsi_per_res_in_rsrv *res;
 		int tmp_len, header_only;
 
 		res = (struct scsi_per_res_in_rsrv *)ctsio->kern_data_ptr;
 
 		scsi_ulto4b(lun->PRGeneration, res->header.generation);
 
 		if (lun->flags & CTL_LUN_PR_RESERVED)
 		{
 			tmp_len = sizeof(struct scsi_per_res_in_rsrv);
 			scsi_ulto4b(sizeof(struct scsi_per_res_in_rsrv_data),
 				    res->header.length);
 			header_only = 0;
 		} else {
 			tmp_len = sizeof(struct scsi_per_res_in_header);
 			scsi_ulto4b(0, res->header.length);
 			header_only = 1;
 		}
 
 		/*
 		 * We had to drop the lock to allocate our buffer, which
 		 * leaves time for someone to come in with another
 		 * persistent reservation.  (That is unlikely, though,
 		 * since this should be the only persistent reservation
 		 * command active right now.)
 		 */
 		if (tmp_len != total_len) {
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			printf("%s: reservation status changed, retrying\n",
 			       __func__);
 			goto retry;
 		}
 
 		/*
 		 * No reservation held, so we're done.
 		 */
 		if (header_only != 0)
 			break;
 
 		/*
 		 * If the registration is an All Registrants type, the key
 		 * is 0, since it doesn't really matter.
 		 */
 		if (lun->pr_res_idx != CTL_PR_ALL_REGISTRANTS) {
 			scsi_u64to8b(ctl_get_prkey(lun, lun->pr_res_idx),
 			    res->data.reservation);
 		}
 		res->data.scopetype = lun->res_type;
 		break;
 	}
 	case SPRI_RC:     //report capabilities
 	{
 		struct scsi_per_res_cap *res_cap;
 		uint16_t type_mask;
 
 		res_cap = (struct scsi_per_res_cap *)ctsio->kern_data_ptr;
 		scsi_ulto2b(sizeof(*res_cap), res_cap->length);
 		res_cap->flags2 |= SPRI_TMV | SPRI_ALLOW_5;
 		type_mask = SPRI_TM_WR_EX_AR |
 			    SPRI_TM_EX_AC_RO |
 			    SPRI_TM_WR_EX_RO |
 			    SPRI_TM_EX_AC |
 			    SPRI_TM_WR_EX |
 			    SPRI_TM_EX_AC_AR;
 		scsi_ulto2b(type_mask, res_cap->type_mask);
 		break;
 	}
 	case SPRI_RS: { // read full status
 		struct scsi_per_res_in_full *res_status;
 		struct scsi_per_res_in_full_desc *res_desc;
 		struct ctl_port *port;
 		int i, len;
 
 		res_status = (struct scsi_per_res_in_full*)ctsio->kern_data_ptr;
 
 		/*
 		 * We had to drop the lock to allocate our buffer, which
 		 * leaves time for someone to come in with another
 		 * persistent reservation.  (That is unlikely, though,
 		 * since this should be the only persistent reservation
 		 * command active right now.)
 		 */
 		if (total_len < (sizeof(struct scsi_per_res_in_header) +
 		    (sizeof(struct scsi_per_res_in_full_desc) + 256) *
 		     lun->pr_key_count)){
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			printf("%s: reservation length changed, retrying\n",
 			       __func__);
 			goto retry;
 		}
 
 		scsi_ulto4b(lun->PRGeneration, res_status->header.generation);
 
 		res_desc = &res_status->desc[0];
 		for (i = 0; i < 2*CTL_MAX_INITIATORS; i++) {
 			if ((key = ctl_get_prkey(lun, i)) == 0)
 				continue;
 
 			scsi_u64to8b(key, res_desc->res_key.key);
 			if ((lun->flags & CTL_LUN_PR_RESERVED) &&
 			    (lun->pr_res_idx == i ||
 			     lun->pr_res_idx == CTL_PR_ALL_REGISTRANTS)) {
 				res_desc->flags = SPRI_FULL_R_HOLDER;
 				res_desc->scopetype = lun->res_type;
 			}
 			scsi_ulto2b(i / CTL_MAX_INIT_PER_PORT,
 			    res_desc->rel_trgt_port_id);
 			len = 0;
 			port = softc->ctl_ports[
 			    ctl_port_idx(i / CTL_MAX_INIT_PER_PORT)];
 			if (port != NULL)
 				len = ctl_create_iid(port,
 				    i % CTL_MAX_INIT_PER_PORT,
 				    res_desc->transport_id);
 			scsi_ulto4b(len, res_desc->additional_length);
 			res_desc = (struct scsi_per_res_in_full_desc *)
 			    &res_desc->transport_id[len];
 		}
 		scsi_ulto4b((uint8_t *)res_desc - (uint8_t *)&res_status->desc[0],
 		    res_status->header.length);
 		break;
 	}
 	default:
 		/*
 		 * This is a bug, because we just checked for this above,
 		 * and should have returned an error.
 		 */
 		panic("Invalid PR type %x", cdb->action);
 		break; /* NOTREACHED */
 	}
 	mtx_unlock(&lun->lun_lock);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static void
 ctl_est_res_ua(struct ctl_lun *lun, uint32_t residx, ctl_ua_type ua)
 {
 	int off = lun->ctl_softc->persis_offset;
 
 	if (residx >= off && residx < off + CTL_MAX_INITIATORS)
 		ctl_est_ua(lun, residx - off, ua);
 }
 
 /*
  * Returns 0 if ctl_persistent_reserve_out() should continue, non-zero if
  * it should return.
  */
 static int
 ctl_pro_preempt(struct ctl_softc *softc, struct ctl_lun *lun, uint64_t res_key,
 		uint64_t sa_res_key, uint8_t type, uint32_t residx,
 		struct ctl_scsiio *ctsio, struct scsi_per_res_out *cdb,
 		struct scsi_per_res_out_parms* param)
 {
 	union ctl_ha_msg persis_io;
 	int retval, i;
 	int isc_retval;
 
 	retval = 0;
 
 	mtx_lock(&lun->lun_lock);
 	if (sa_res_key == 0) {
 		if (lun->pr_res_idx == CTL_PR_ALL_REGISTRANTS) {
 			/* validate scope and type */
 			if ((cdb->scope_type & SPR_SCOPE_MASK) !=
 			     SPR_LU_SCOPE) {
 				mtx_unlock(&lun->lun_lock);
 				ctl_set_invalid_field(/*ctsio*/ ctsio,
 						      /*sks_valid*/ 1,
 						      /*command*/ 1,
 						      /*field*/ 2,
 						      /*bit_valid*/ 1,
 						      /*bit*/ 4);
 				ctl_done((union ctl_io *)ctsio);
 				return (1);
 			}
 
 		        if (type>8 || type==2 || type==4 || type==0) {
 				mtx_unlock(&lun->lun_lock);
 				ctl_set_invalid_field(/*ctsio*/ ctsio,
        	           				      /*sks_valid*/ 1,
 						      /*command*/ 1,
 						      /*field*/ 2,
 						      /*bit_valid*/ 1,
 						      /*bit*/ 0);
 				ctl_done((union ctl_io *)ctsio);
 				return (1);
 		        }
 
 			/*
 			 * Unregister everybody else and build UA for
 			 * them
 			 */
 			for(i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 				if (i == residx || ctl_get_prkey(lun, i) == 0)
 					continue;
 
 				ctl_clr_prkey(lun, i);
 				ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 			}
 			lun->pr_key_count = 1;
 			lun->res_type = type;
 			if (lun->res_type != SPR_TYPE_WR_EX_AR
 			 && lun->res_type != SPR_TYPE_EX_AC_AR)
 				lun->pr_res_idx = residx;
 
 			/* send msg to other side */
 			persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 			persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 			persis_io.pr.pr_info.action = CTL_PR_PREEMPT;
 			persis_io.pr.pr_info.residx = lun->pr_res_idx;
 			persis_io.pr.pr_info.res_type = type;
 			memcpy(persis_io.pr.pr_info.sa_res_key,
 			       param->serv_act_res_key,
 			       sizeof(param->serv_act_res_key));
 			if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			     &persis_io, sizeof(persis_io), 0)) >
 			     CTL_HA_STATUS_SUCCESS) {
 				printf("CTL:Persis Out error returned "
 				       "from ctl_ha_msg_send %d\n",
 				       isc_retval);
 			}
 		} else {
 			/* not all registrants */
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 0,
 					      /*field*/ 8,
 					      /*bit_valid*/ 0,
 					      /*bit*/ 0);
 			ctl_done((union ctl_io *)ctsio);
 			return (1);
 		}
 	} else if (lun->pr_res_idx == CTL_PR_ALL_REGISTRANTS
 		|| !(lun->flags & CTL_LUN_PR_RESERVED)) {
 		int found = 0;
 
 		if (res_key == sa_res_key) {
 			/* special case */
 			/*
 			 * The spec implies this is not good but doesn't
 			 * say what to do. There are two choices either
 			 * generate a res conflict or check condition
 			 * with illegal field in parameter data. Since
 			 * that is what is done when the sa_res_key is
 			 * zero I'll take that approach since this has
 			 * to do with the sa_res_key.
 			 */
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 0,
 					      /*field*/ 8,
 					      /*bit_valid*/ 0,
 					      /*bit*/ 0);
 			ctl_done((union ctl_io *)ctsio);
 			return (1);
 		}
 
 		for (i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 			if (ctl_get_prkey(lun, i) != sa_res_key)
 				continue;
 
 			found = 1;
 			ctl_clr_prkey(lun, i);
 			lun->pr_key_count--;
 			ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 		}
 		if (!found) {
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_reservation_conflict(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		/* send msg to other side */
 		persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 		persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 		persis_io.pr.pr_info.action = CTL_PR_PREEMPT;
 		persis_io.pr.pr_info.residx = lun->pr_res_idx;
 		persis_io.pr.pr_info.res_type = type;
 		memcpy(persis_io.pr.pr_info.sa_res_key,
 		       param->serv_act_res_key,
 		       sizeof(param->serv_act_res_key));
 		if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 		     &persis_io, sizeof(persis_io), 0)) >
 		     CTL_HA_STATUS_SUCCESS) {
 			printf("CTL:Persis Out error returned from "
 			       "ctl_ha_msg_send %d\n", isc_retval);
 		}
 	} else {
 		/* Reserved but not all registrants */
 		/* sa_res_key is res holder */
 		if (sa_res_key == ctl_get_prkey(lun, lun->pr_res_idx)) {
 			/* validate scope and type */
 			if ((cdb->scope_type & SPR_SCOPE_MASK) !=
 			     SPR_LU_SCOPE) {
 				mtx_unlock(&lun->lun_lock);
 				ctl_set_invalid_field(/*ctsio*/ ctsio,
 						      /*sks_valid*/ 1,
 						      /*command*/ 1,
 						      /*field*/ 2,
 						      /*bit_valid*/ 1,
 						      /*bit*/ 4);
 				ctl_done((union ctl_io *)ctsio);
 				return (1);
 			}
 
 			if (type>8 || type==2 || type==4 || type==0) {
 				mtx_unlock(&lun->lun_lock);
 				ctl_set_invalid_field(/*ctsio*/ ctsio,
 						      /*sks_valid*/ 1,
 						      /*command*/ 1,
 						      /*field*/ 2,
 						      /*bit_valid*/ 1,
 						      /*bit*/ 0);
 				ctl_done((union ctl_io *)ctsio);
 				return (1);
 			}
 
 			/*
 			 * Do the following:
 			 * if sa_res_key != res_key remove all
 			 * registrants w/sa_res_key and generate UA
 			 * for these registrants(Registrations
 			 * Preempted) if it wasn't an exclusive
 			 * reservation generate UA(Reservations
 			 * Preempted) for all other registered nexuses
 			 * if the type has changed. Establish the new
 			 * reservation and holder. If res_key and
 			 * sa_res_key are the same do the above
 			 * except don't unregister the res holder.
 			 */
 
 			for(i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 				if (i == residx || ctl_get_prkey(lun, i) == 0)
 					continue;
 
 				if (sa_res_key == ctl_get_prkey(lun, i)) {
 					ctl_clr_prkey(lun, i);
 					lun->pr_key_count--;
 					ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 				} else if (type != lun->res_type
 					&& (lun->res_type == SPR_TYPE_WR_EX_RO
 					 || lun->res_type ==SPR_TYPE_EX_AC_RO)){
 					ctl_est_res_ua(lun, i, CTL_UA_RES_RELEASE);
 				}
 			}
 			lun->res_type = type;
 			if (lun->res_type != SPR_TYPE_WR_EX_AR
 			 && lun->res_type != SPR_TYPE_EX_AC_AR)
 				lun->pr_res_idx = residx;
 			else
 				lun->pr_res_idx = CTL_PR_ALL_REGISTRANTS;
 
 			persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 			persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 			persis_io.pr.pr_info.action = CTL_PR_PREEMPT;
 			persis_io.pr.pr_info.residx = lun->pr_res_idx;
 			persis_io.pr.pr_info.res_type = type;
 			memcpy(persis_io.pr.pr_info.sa_res_key,
 			       param->serv_act_res_key,
 			       sizeof(param->serv_act_res_key));
 			if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			     &persis_io, sizeof(persis_io), 0)) >
 			     CTL_HA_STATUS_SUCCESS) {
 				printf("CTL:Persis Out error returned "
 				       "from ctl_ha_msg_send %d\n",
 				       isc_retval);
 			}
 		} else {
 			/*
 			 * sa_res_key is not the res holder just
 			 * remove registrants
 			 */
 			int found=0;
 
 			for (i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 				if (sa_res_key != ctl_get_prkey(lun, i))
 					continue;
 
 				found = 1;
 				ctl_clr_prkey(lun, i);
 				lun->pr_key_count--;
 				ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 			}
 
 			if (!found) {
 				mtx_unlock(&lun->lun_lock);
 				free(ctsio->kern_data_ptr, M_CTL);
 				ctl_set_reservation_conflict(ctsio);
 				ctl_done((union ctl_io *)ctsio);
 		        	return (1);
 			}
 			persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 			persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 			persis_io.pr.pr_info.action = CTL_PR_PREEMPT;
 			persis_io.pr.pr_info.residx = lun->pr_res_idx;
 			persis_io.pr.pr_info.res_type = type;
 			memcpy(persis_io.pr.pr_info.sa_res_key,
 			       param->serv_act_res_key,
 			       sizeof(param->serv_act_res_key));
 			if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			     &persis_io, sizeof(persis_io), 0)) >
 			     CTL_HA_STATUS_SUCCESS) {
 				printf("CTL:Persis Out error returned "
 				       "from ctl_ha_msg_send %d\n",
 				isc_retval);
 			}
 		}
 	}
 
 	lun->PRGeneration++;
 	mtx_unlock(&lun->lun_lock);
 
 	return (retval);
 }
 
 static void
 ctl_pro_preempt_other(struct ctl_lun *lun, union ctl_ha_msg *msg)
 {
 	uint64_t sa_res_key;
 	int i;
 
 	sa_res_key = scsi_8btou64(msg->pr.pr_info.sa_res_key);
 
 	if (lun->pr_res_idx == CTL_PR_ALL_REGISTRANTS
 	 || lun->pr_res_idx == CTL_PR_NO_RESERVATION
 	 || sa_res_key != ctl_get_prkey(lun, lun->pr_res_idx)) {
 		if (sa_res_key == 0) {
 			/*
 			 * Unregister everybody else and build UA for
 			 * them
 			 */
 			for(i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 				if (i == msg->pr.pr_info.residx ||
 				    ctl_get_prkey(lun, i) == 0)
 					continue;
 
 				ctl_clr_prkey(lun, i);
 				ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 			}
 
 			lun->pr_key_count = 1;
 			lun->res_type = msg->pr.pr_info.res_type;
 			if (lun->res_type != SPR_TYPE_WR_EX_AR
 			 && lun->res_type != SPR_TYPE_EX_AC_AR)
 				lun->pr_res_idx = msg->pr.pr_info.residx;
 		} else {
 		        for (i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 				if (sa_res_key == ctl_get_prkey(lun, i))
 					continue;
 
 				ctl_clr_prkey(lun, i);
 				lun->pr_key_count--;
 				ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 			}
 		}
 	} else {
 		for (i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 			if (i == msg->pr.pr_info.residx ||
 			    ctl_get_prkey(lun, i) == 0)
 				continue;
 
 			if (sa_res_key == ctl_get_prkey(lun, i)) {
 				ctl_clr_prkey(lun, i);
 				lun->pr_key_count--;
 				ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 			} else if (msg->pr.pr_info.res_type != lun->res_type
 				&& (lun->res_type == SPR_TYPE_WR_EX_RO
 				 || lun->res_type == SPR_TYPE_EX_AC_RO)) {
 				ctl_est_res_ua(lun, i, CTL_UA_RES_RELEASE);
 			}
 		}
 		lun->res_type = msg->pr.pr_info.res_type;
 		if (lun->res_type != SPR_TYPE_WR_EX_AR
 		 && lun->res_type != SPR_TYPE_EX_AC_AR)
 			lun->pr_res_idx = msg->pr.pr_info.residx;
 		else
 			lun->pr_res_idx = CTL_PR_ALL_REGISTRANTS;
 	}
 	lun->PRGeneration++;
 
 }
 
 
 int
 ctl_persistent_reserve_out(struct ctl_scsiio *ctsio)
 {
 	int retval;
 	int isc_retval;
 	u_int32_t param_len;
 	struct scsi_per_res_out *cdb;
 	struct ctl_lun *lun;
 	struct scsi_per_res_out_parms* param;
 	struct ctl_softc *softc;
 	uint32_t residx;
 	uint64_t res_key, sa_res_key, key;
 	uint8_t type;
 	union ctl_ha_msg persis_io;
 	int    i;
 
 	CTL_DEBUG_PRINT(("ctl_persistent_reserve_out\n"));
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	cdb = (struct scsi_per_res_out *)ctsio->cdb;
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	softc = lun->ctl_softc;
 
 	/*
 	 * We only support whole-LUN scope.  The scope & type are ignored for
 	 * register, register and ignore existing key and clear.
 	 * We sometimes ignore scope and type on preempts too!!
 	 * Verify reservation type here as well.
 	 */
 	type = cdb->scope_type & SPR_TYPE_MASK;
 	if ((cdb->action == SPRO_RESERVE)
 	 || (cdb->action == SPRO_RELEASE)) {
 		if ((cdb->scope_type & SPR_SCOPE_MASK) != SPR_LU_SCOPE) {
 			ctl_set_invalid_field(/*ctsio*/ ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 2,
 					      /*bit_valid*/ 1,
 					      /*bit*/ 4);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 
 		if (type>8 || type==2 || type==4 || type==0) {
 			ctl_set_invalid_field(/*ctsio*/ ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 2,
 					      /*bit_valid*/ 1,
 					      /*bit*/ 0);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 	}
 
 	param_len = scsi_4btoul(cdb->length);
 
 	if ((ctsio->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0) {
 		ctsio->kern_data_ptr = malloc(param_len, M_CTL, M_WAITOK);
 		ctsio->kern_data_len = param_len;
 		ctsio->kern_total_len = param_len;
 		ctsio->kern_data_resid = 0;
 		ctsio->kern_rel_offset = 0;
 		ctsio->kern_sg_entries = 0;
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	param = (struct scsi_per_res_out_parms *)ctsio->kern_data_ptr;
 
 	residx = ctl_get_resindex(&ctsio->io_hdr.nexus);
 	res_key = scsi_8btou64(param->res_key.key);
 	sa_res_key = scsi_8btou64(param->serv_act_res_key);
 
 	/*
 	 * Validate the reservation key here except for SPRO_REG_IGNO
 	 * This must be done for all other service actions
 	 */
 	if ((cdb->action & SPRO_ACTION_MASK) != SPRO_REG_IGNO) {
 		mtx_lock(&lun->lun_lock);
 		if ((key = ctl_get_prkey(lun, residx)) != 0) {
 			if (res_key != key) {
 				/*
 				 * The current key passed in doesn't match
 				 * the one the initiator previously
 				 * registered.
 				 */
 				mtx_unlock(&lun->lun_lock);
 				free(ctsio->kern_data_ptr, M_CTL);
 				ctl_set_reservation_conflict(ctsio);
 				ctl_done((union ctl_io *)ctsio);
 				return (CTL_RETVAL_COMPLETE);
 			}
 		} else if ((cdb->action & SPRO_ACTION_MASK) != SPRO_REGISTER) {
 			/*
 			 * We are not registered
 			 */
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_reservation_conflict(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		} else if (res_key != 0) {
 			/*
 			 * We are not registered and trying to register but
 			 * the register key isn't zero.
 			 */
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_reservation_conflict(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		mtx_unlock(&lun->lun_lock);
 	}
 
 	switch (cdb->action & SPRO_ACTION_MASK) {
 	case SPRO_REGISTER:
 	case SPRO_REG_IGNO: {
 
 #if 0
 		printf("Registration received\n");
 #endif
 
 		/*
 		 * We don't support any of these options, as we report in
 		 * the read capabilities request (see
 		 * ctl_persistent_reserve_in(), above).
 		 */
 		if ((param->flags & SPR_SPEC_I_PT)
 		 || (param->flags & SPR_ALL_TG_PT)
 		 || (param->flags & SPR_APTPL)) {
 			int bit_ptr;
 
 			if (param->flags & SPR_APTPL)
 				bit_ptr = 0;
 			else if (param->flags & SPR_ALL_TG_PT)
 				bit_ptr = 2;
 			else /* SPR_SPEC_I_PT */
 				bit_ptr = 3;
 
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 0,
 					      /*field*/ 20,
 					      /*bit_valid*/ 1,
 					      /*bit*/ bit_ptr);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 
 		mtx_lock(&lun->lun_lock);
 
 		/*
 		 * The initiator wants to clear the
 		 * key/unregister.
 		 */
 		if (sa_res_key == 0) {
 			if ((res_key == 0
 			  && (cdb->action & SPRO_ACTION_MASK) == SPRO_REGISTER)
 			 || ((cdb->action & SPRO_ACTION_MASK) == SPRO_REG_IGNO
 			  && ctl_get_prkey(lun, residx) == 0)) {
 				mtx_unlock(&lun->lun_lock);
 				goto done;
 			}
 
 			ctl_clr_prkey(lun, residx);
 			lun->pr_key_count--;
 
 			if (residx == lun->pr_res_idx) {
 				lun->flags &= ~CTL_LUN_PR_RESERVED;
 				lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 
 				if ((lun->res_type == SPR_TYPE_WR_EX_RO
 				  || lun->res_type == SPR_TYPE_EX_AC_RO)
 				 && lun->pr_key_count) {
 					/*
 					 * If the reservation is a registrants
 					 * only type we need to generate a UA
 					 * for other registered inits.  The
 					 * sense code should be RESERVATIONS
 					 * RELEASED
 					 */
 
 					for (i = 0; i < CTL_MAX_INITIATORS;i++){
 						if (ctl_get_prkey(lun, i +
 						    softc->persis_offset) == 0)
 							continue;
 						ctl_est_ua(lun, i,
 						    CTL_UA_RES_RELEASE);
 					}
 				}
 				lun->res_type = 0;
 			} else if (lun->pr_res_idx == CTL_PR_ALL_REGISTRANTS) {
 				if (lun->pr_key_count==0) {
 					lun->flags &= ~CTL_LUN_PR_RESERVED;
 					lun->res_type = 0;
 					lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 				}
 			}
 			persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 			persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 			persis_io.pr.pr_info.action = CTL_PR_UNREG_KEY;
 			persis_io.pr.pr_info.residx = residx;
 			if ((isc_retval = ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			     &persis_io, sizeof(persis_io), 0 )) >
 			     CTL_HA_STATUS_SUCCESS) {
 				printf("CTL:Persis Out error returned from "
 				       "ctl_ha_msg_send %d\n", isc_retval);
 			}
 		} else /* sa_res_key != 0 */ {
 
 			/*
 			 * If we aren't registered currently then increment
 			 * the key count and set the registered flag.
 			 */
 			ctl_alloc_prkey(lun, residx);
 			if (ctl_get_prkey(lun, residx) == 0)
 				lun->pr_key_count++;
 			ctl_set_prkey(lun, residx, sa_res_key);
 
 			persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 			persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 			persis_io.pr.pr_info.action = CTL_PR_REG_KEY;
 			persis_io.pr.pr_info.residx = residx;
 			memcpy(persis_io.pr.pr_info.sa_res_key,
 			       param->serv_act_res_key,
 			       sizeof(param->serv_act_res_key));
 			if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			     &persis_io, sizeof(persis_io), 0)) >
 			     CTL_HA_STATUS_SUCCESS) {
 				printf("CTL:Persis Out error returned from "
 				       "ctl_ha_msg_send %d\n", isc_retval);
 			}
 		}
 		lun->PRGeneration++;
 		mtx_unlock(&lun->lun_lock);
 
 		break;
 	}
 	case SPRO_RESERVE:
 #if 0
                 printf("Reserve executed type %d\n", type);
 #endif
 		mtx_lock(&lun->lun_lock);
 		if (lun->flags & CTL_LUN_PR_RESERVED) {
 			/*
 			 * if this isn't the reservation holder and it's
 			 * not a "all registrants" type or if the type is
 			 * different then we have a conflict
 			 */
 			if ((lun->pr_res_idx != residx
 			  && lun->pr_res_idx != CTL_PR_ALL_REGISTRANTS)
 			 || lun->res_type != type) {
 				mtx_unlock(&lun->lun_lock);
 				free(ctsio->kern_data_ptr, M_CTL);
 				ctl_set_reservation_conflict(ctsio);
 				ctl_done((union ctl_io *)ctsio);
 				return (CTL_RETVAL_COMPLETE);
 			}
 			mtx_unlock(&lun->lun_lock);
 		} else /* create a reservation */ {
 			/*
 			 * If it's not an "all registrants" type record
 			 * reservation holder
 			 */
 			if (type != SPR_TYPE_WR_EX_AR
 			 && type != SPR_TYPE_EX_AC_AR)
 				lun->pr_res_idx = residx; /* Res holder */
 			else
 				lun->pr_res_idx = CTL_PR_ALL_REGISTRANTS;
 
 			lun->flags |= CTL_LUN_PR_RESERVED;
 			lun->res_type = type;
 
 			mtx_unlock(&lun->lun_lock);
 
 			/* send msg to other side */
 			persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 			persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 			persis_io.pr.pr_info.action = CTL_PR_RESERVE;
 			persis_io.pr.pr_info.residx = lun->pr_res_idx;
 			persis_io.pr.pr_info.res_type = type;
 			if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			     &persis_io, sizeof(persis_io), 0)) >
 			     CTL_HA_STATUS_SUCCESS) {
 				printf("CTL:Persis Out error returned from "
 				       "ctl_ha_msg_send %d\n", isc_retval);
 			}
 		}
 		break;
 
 	case SPRO_RELEASE:
 		mtx_lock(&lun->lun_lock);
 		if ((lun->flags & CTL_LUN_PR_RESERVED) == 0) {
 			/* No reservation exists return good status */
 			mtx_unlock(&lun->lun_lock);
 			goto done;
 		}
 		/*
 		 * Is this nexus a reservation holder?
 		 */
 		if (lun->pr_res_idx != residx
 		 && lun->pr_res_idx != CTL_PR_ALL_REGISTRANTS) {
 			/*
 			 * not a res holder return good status but
 			 * do nothing
 			 */
 			mtx_unlock(&lun->lun_lock);
 			goto done;
 		}
 
 		if (lun->res_type != type) {
 			mtx_unlock(&lun->lun_lock);
 			free(ctsio->kern_data_ptr, M_CTL);
 			ctl_set_illegal_pr_release(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 
 		/* okay to release */
 		lun->flags &= ~CTL_LUN_PR_RESERVED;
 		lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 		lun->res_type = 0;
 
 		/*
 		 * if this isn't an exclusive access
 		 * res generate UA for all other
 		 * registrants.
 		 */
 		if (type != SPR_TYPE_EX_AC
 		 && type != SPR_TYPE_WR_EX) {
 			for (i = 0; i < CTL_MAX_INITIATORS; i++) {
 				if (i == residx ||
 				    ctl_get_prkey(lun,
 				     i + softc->persis_offset) == 0)
 					continue;
 				ctl_est_ua(lun, i, CTL_UA_RES_RELEASE);
 			}
 		}
 		mtx_unlock(&lun->lun_lock);
 		/* Send msg to other side */
 		persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 		persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 		persis_io.pr.pr_info.action = CTL_PR_RELEASE;
 		if ((isc_retval=ctl_ha_msg_send( CTL_HA_CHAN_CTL, &persis_io,
 		     sizeof(persis_io), 0)) > CTL_HA_STATUS_SUCCESS) {
 			printf("CTL:Persis Out error returned from "
 			       "ctl_ha_msg_send %d\n", isc_retval);
 		}
 		break;
 
 	case SPRO_CLEAR:
 		/* send msg to other side */
 
 		mtx_lock(&lun->lun_lock);
 		lun->flags &= ~CTL_LUN_PR_RESERVED;
 		lun->res_type = 0;
 		lun->pr_key_count = 0;
 		lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 
 		ctl_clr_prkey(lun, residx);
 		for (i=0; i < 2*CTL_MAX_INITIATORS; i++)
 			if (ctl_get_prkey(lun, i) != 0) {
 				ctl_clr_prkey(lun, i);
 				ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 			}
 		lun->PRGeneration++;
 		mtx_unlock(&lun->lun_lock);
 		persis_io.hdr.nexus = ctsio->io_hdr.nexus;
 		persis_io.hdr.msg_type = CTL_MSG_PERS_ACTION;
 		persis_io.pr.pr_info.action = CTL_PR_CLEAR;
 		if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL, &persis_io,
 		     sizeof(persis_io), 0)) > CTL_HA_STATUS_SUCCESS) {
 			printf("CTL:Persis Out error returned from "
 			       "ctl_ha_msg_send %d\n", isc_retval);
 		}
 		break;
 
 	case SPRO_PREEMPT:
 	case SPRO_PRE_ABO: {
 		int nretval;
 
 		nretval = ctl_pro_preempt(softc, lun, res_key, sa_res_key, type,
 					  residx, ctsio, cdb, param);
 		if (nretval != 0)
 			return (CTL_RETVAL_COMPLETE);
 		break;
 	}
 	default:
 		panic("Invalid PR type %x", cdb->action);
 	}
 
 done:
 	free(ctsio->kern_data_ptr, M_CTL);
 	ctl_set_success(ctsio);
 	ctl_done((union ctl_io *)ctsio);
 
 	return (retval);
 }
 
 /*
  * This routine is for handling a message from the other SC pertaining to
  * persistent reserve out. All the error checking will have been done
  * so only perorming the action need be done here to keep the two
  * in sync.
  */
 static void
 ctl_hndl_per_res_out_on_other_sc(union ctl_ha_msg *msg)
 {
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	int i;
 	uint32_t targ_lun;
 
 	softc = control_softc;
 
 	targ_lun = msg->hdr.nexus.targ_mapped_lun;
 	lun = softc->ctl_luns[targ_lun];
 	mtx_lock(&lun->lun_lock);
 	switch(msg->pr.pr_info.action) {
 	case CTL_PR_REG_KEY:
 		ctl_alloc_prkey(lun, msg->pr.pr_info.residx);
 		if (ctl_get_prkey(lun, msg->pr.pr_info.residx) == 0)
 			lun->pr_key_count++;
 		ctl_set_prkey(lun, msg->pr.pr_info.residx,
 		    scsi_8btou64(msg->pr.pr_info.sa_res_key));
 		lun->PRGeneration++;
 		break;
 
 	case CTL_PR_UNREG_KEY:
 		ctl_clr_prkey(lun, msg->pr.pr_info.residx);
 		lun->pr_key_count--;
 
 		/* XXX Need to see if the reservation has been released */
 		/* if so do we need to generate UA? */
 		if (msg->pr.pr_info.residx == lun->pr_res_idx) {
 			lun->flags &= ~CTL_LUN_PR_RESERVED;
 			lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 
 			if ((lun->res_type == SPR_TYPE_WR_EX_RO
 			  || lun->res_type == SPR_TYPE_EX_AC_RO)
 			 && lun->pr_key_count) {
 				/*
 				 * If the reservation is a registrants
 				 * only type we need to generate a UA
 				 * for other registered inits.  The
 				 * sense code should be RESERVATIONS
 				 * RELEASED
 				 */
 
 				for (i = 0; i < CTL_MAX_INITIATORS; i++) {
 					if (ctl_get_prkey(lun, i +
 					    softc->persis_offset) == 0)
 						continue;
 
 					ctl_est_ua(lun, i, CTL_UA_RES_RELEASE);
 				}
 			}
 			lun->res_type = 0;
 		} else if (lun->pr_res_idx == CTL_PR_ALL_REGISTRANTS) {
 			if (lun->pr_key_count==0) {
 				lun->flags &= ~CTL_LUN_PR_RESERVED;
 				lun->res_type = 0;
 				lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 			}
 		}
 		lun->PRGeneration++;
 		break;
 
 	case CTL_PR_RESERVE:
 		lun->flags |= CTL_LUN_PR_RESERVED;
 		lun->res_type = msg->pr.pr_info.res_type;
 		lun->pr_res_idx = msg->pr.pr_info.residx;
 
 		break;
 
 	case CTL_PR_RELEASE:
 		/*
 		 * if this isn't an exclusive access res generate UA for all
 		 * other registrants.
 		 */
 		if (lun->res_type != SPR_TYPE_EX_AC
 		 && lun->res_type != SPR_TYPE_WR_EX) {
 			for (i = 0; i < CTL_MAX_INITIATORS; i++)
 				if (ctl_get_prkey(lun, i + softc->persis_offset) != 0)
 					ctl_est_ua(lun, i, CTL_UA_RES_RELEASE);
 		}
 
 		lun->flags &= ~CTL_LUN_PR_RESERVED;
 		lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 		lun->res_type = 0;
 		break;
 
 	case CTL_PR_PREEMPT:
 		ctl_pro_preempt_other(lun, msg);
 		break;
 	case CTL_PR_CLEAR:
 		lun->flags &= ~CTL_LUN_PR_RESERVED;
 		lun->res_type = 0;
 		lun->pr_key_count = 0;
 		lun->pr_res_idx = CTL_PR_NO_RESERVATION;
 
 		for (i=0; i < 2*CTL_MAX_INITIATORS; i++) {
 			if (ctl_get_prkey(lun, i) == 0)
 				continue;
 			ctl_clr_prkey(lun, i);
 			ctl_est_res_ua(lun, i, CTL_UA_REG_PREEMPT);
 		}
 		lun->PRGeneration++;
 		break;
 	}
 
 	mtx_unlock(&lun->lun_lock);
 }
 
 int
 ctl_read_write(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct ctl_lba_len_flags *lbalen;
 	uint64_t lba;
 	uint32_t num_blocks;
 	int flags, retval;
 	int isread;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	CTL_DEBUG_PRINT(("ctl_read_write: command: %#x\n", ctsio->cdb[0]));
 
 	flags = 0;
 	retval = CTL_RETVAL_COMPLETE;
 
 	isread = ctsio->cdb[0] == READ_6  || ctsio->cdb[0] == READ_10
 	      || ctsio->cdb[0] == READ_12 || ctsio->cdb[0] == READ_16;
 	switch (ctsio->cdb[0]) {
 	case READ_6:
 	case WRITE_6: {
 		struct scsi_rw_6 *cdb;
 
 		cdb = (struct scsi_rw_6 *)ctsio->cdb;
 
 		lba = scsi_3btoul(cdb->addr);
 		/* only 5 bits are valid in the most significant address byte */
 		lba &= 0x1fffff;
 		num_blocks = cdb->length;
 		/*
 		 * This is correct according to SBC-2.
 		 */
 		if (num_blocks == 0)
 			num_blocks = 256;
 		break;
 	}
 	case READ_10:
 	case WRITE_10: {
 		struct scsi_rw_10 *cdb;
 
 		cdb = (struct scsi_rw_10 *)ctsio->cdb;
 		if (cdb->byte2 & SRW10_FUA)
 			flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SRW10_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_2btoul(cdb->length);
 		break;
 	}
 	case WRITE_VERIFY_10: {
 		struct scsi_write_verify_10 *cdb;
 
 		cdb = (struct scsi_write_verify_10 *)ctsio->cdb;
 		flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SWV_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_2btoul(cdb->length);
 		break;
 	}
 	case READ_12:
 	case WRITE_12: {
 		struct scsi_rw_12 *cdb;
 
 		cdb = (struct scsi_rw_12 *)ctsio->cdb;
 		if (cdb->byte2 & SRW12_FUA)
 			flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SRW12_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		break;
 	}
 	case WRITE_VERIFY_12: {
 		struct scsi_write_verify_12 *cdb;
 
 		cdb = (struct scsi_write_verify_12 *)ctsio->cdb;
 		flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SWV_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		break;
 	}
 	case READ_16:
 	case WRITE_16: {
 		struct scsi_rw_16 *cdb;
 
 		cdb = (struct scsi_rw_16 *)ctsio->cdb;
 		if (cdb->byte2 & SRW12_FUA)
 			flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SRW12_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_8btou64(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		break;
 	}
 	case WRITE_ATOMIC_16: {
 		struct scsi_rw_16 *cdb;
 
 		if (lun->be_lun->atomicblock == 0) {
 			ctl_set_invalid_opcode(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 
 		cdb = (struct scsi_rw_16 *)ctsio->cdb;
 		if (cdb->byte2 & SRW12_FUA)
 			flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SRW12_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_8btou64(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		if (num_blocks > lun->be_lun->atomicblock) {
 			ctl_set_invalid_field(ctsio, /*sks_valid*/ 1,
 			    /*command*/ 1, /*field*/ 12, /*bit_valid*/ 0,
 			    /*bit*/ 0);
 			ctl_done((union ctl_io *)ctsio);
 			return (CTL_RETVAL_COMPLETE);
 		}
 		break;
 	}
 	case WRITE_VERIFY_16: {
 		struct scsi_write_verify_16 *cdb;
 
 		cdb = (struct scsi_write_verify_16 *)ctsio->cdb;
 		flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SWV_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_8btou64(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		break;
 	}
 	default:
 		/*
 		 * We got a command we don't support.  This shouldn't
 		 * happen, commands should be filtered out above us.
 		 */
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 		break; /* NOTREACHED */
 	}
 
 	/*
 	 * The first check is to make sure we're in bounds, the second
 	 * check is to catch wrap-around problems.  If the lba + num blocks
 	 * is less than the lba, then we've wrapped around and the block
 	 * range is invalid anyway.
 	 */
 	if (((lba + num_blocks) > (lun->be_lun->maxlba + 1))
 	 || ((lba + num_blocks) < lba)) {
 		ctl_set_lba_out_of_range(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * According to SBC-3, a transfer length of 0 is not an error.
 	 * Note that this cannot happen with WRITE(6) or READ(6), since 0
 	 * translates to 256 blocks for those commands.
 	 */
 	if (num_blocks == 0) {
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/* Set FUA and/or DPO if caches are disabled. */
 	if (isread) {
 		if ((lun->mode_pages.caching_page[CTL_PAGE_CURRENT].flags1 &
 		    SCP_RCD) != 0)
 			flags |= CTL_LLF_FUA | CTL_LLF_DPO;
 	} else {
 		if ((lun->mode_pages.caching_page[CTL_PAGE_CURRENT].flags1 &
 		    SCP_WCE) == 0)
 			flags |= CTL_LLF_FUA;
 	}
 
 	lbalen = (struct ctl_lba_len_flags *)
 	    &ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	lbalen->lba = lba;
 	lbalen->len = num_blocks;
 	lbalen->flags = (isread ? CTL_LLF_READ : CTL_LLF_WRITE) | flags;
 
 	ctsio->kern_total_len = num_blocks * lun->be_lun->blocksize;
 	ctsio->kern_rel_offset = 0;
 
 	CTL_DEBUG_PRINT(("ctl_read_write: calling data_submit()\n"));
 
 	retval = lun->backend->data_submit((union ctl_io *)ctsio);
 
 	return (retval);
 }
 
 static int
 ctl_cnw_cont(union ctl_io *io)
 {
 	struct ctl_scsiio *ctsio;
 	struct ctl_lun *lun;
 	struct ctl_lba_len_flags *lbalen;
 	int retval;
 
 	ctsio = &io->scsiio;
 	ctsio->io_hdr.status = CTL_STATUS_NONE;
 	ctsio->io_hdr.flags &= ~CTL_FLAG_IO_CONT;
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	lbalen = (struct ctl_lba_len_flags *)
 	    &ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	lbalen->flags &= ~CTL_LLF_COMPARE;
 	lbalen->flags |= CTL_LLF_WRITE;
 
 	CTL_DEBUG_PRINT(("ctl_cnw_cont: calling data_submit()\n"));
 	retval = lun->backend->data_submit((union ctl_io *)ctsio);
 	return (retval);
 }
 
 int
 ctl_cnw(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct ctl_lba_len_flags *lbalen;
 	uint64_t lba;
 	uint32_t num_blocks;
 	int flags, retval;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	CTL_DEBUG_PRINT(("ctl_cnw: command: %#x\n", ctsio->cdb[0]));
 
 	flags = 0;
 	retval = CTL_RETVAL_COMPLETE;
 
 	switch (ctsio->cdb[0]) {
 	case COMPARE_AND_WRITE: {
 		struct scsi_compare_and_write *cdb;
 
 		cdb = (struct scsi_compare_and_write *)ctsio->cdb;
 		if (cdb->byte2 & SRW10_FUA)
 			flags |= CTL_LLF_FUA;
 		if (cdb->byte2 & SRW10_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_8btou64(cdb->addr);
 		num_blocks = cdb->length;
 		break;
 	}
 	default:
 		/*
 		 * We got a command we don't support.  This shouldn't
 		 * happen, commands should be filtered out above us.
 		 */
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 
 		return (CTL_RETVAL_COMPLETE);
 		break; /* NOTREACHED */
 	}
 
 	/*
 	 * The first check is to make sure we're in bounds, the second
 	 * check is to catch wrap-around problems.  If the lba + num blocks
 	 * is less than the lba, then we've wrapped around and the block
 	 * range is invalid anyway.
 	 */
 	if (((lba + num_blocks) > (lun->be_lun->maxlba + 1))
 	 || ((lba + num_blocks) < lba)) {
 		ctl_set_lba_out_of_range(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * According to SBC-3, a transfer length of 0 is not an error.
 	 */
 	if (num_blocks == 0) {
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/* Set FUA if write cache is disabled. */
 	if ((lun->mode_pages.caching_page[CTL_PAGE_CURRENT].flags1 &
 	    SCP_WCE) == 0)
 		flags |= CTL_LLF_FUA;
 
 	ctsio->kern_total_len = 2 * num_blocks * lun->be_lun->blocksize;
 	ctsio->kern_rel_offset = 0;
 
 	/*
 	 * Set the IO_CONT flag, so that if this I/O gets passed to
 	 * ctl_data_submit_done(), it'll get passed back to
 	 * ctl_ctl_cnw_cont() for further processing.
 	 */
 	ctsio->io_hdr.flags |= CTL_FLAG_IO_CONT;
 	ctsio->io_cont = ctl_cnw_cont;
 
 	lbalen = (struct ctl_lba_len_flags *)
 	    &ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	lbalen->lba = lba;
 	lbalen->len = num_blocks;
 	lbalen->flags = CTL_LLF_COMPARE | flags;
 
 	CTL_DEBUG_PRINT(("ctl_cnw: calling data_submit()\n"));
 	retval = lun->backend->data_submit((union ctl_io *)ctsio);
 	return (retval);
 }
 
 int
 ctl_verify(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct ctl_lba_len_flags *lbalen;
 	uint64_t lba;
 	uint32_t num_blocks;
 	int bytchk, flags;
 	int retval;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	CTL_DEBUG_PRINT(("ctl_verify: command: %#x\n", ctsio->cdb[0]));
 
 	bytchk = 0;
 	flags = CTL_LLF_FUA;
 	retval = CTL_RETVAL_COMPLETE;
 
 	switch (ctsio->cdb[0]) {
 	case VERIFY_10: {
 		struct scsi_verify_10 *cdb;
 
 		cdb = (struct scsi_verify_10 *)ctsio->cdb;
 		if (cdb->byte2 & SVFY_BYTCHK)
 			bytchk = 1;
 		if (cdb->byte2 & SVFY_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_2btoul(cdb->length);
 		break;
 	}
 	case VERIFY_12: {
 		struct scsi_verify_12 *cdb;
 
 		cdb = (struct scsi_verify_12 *)ctsio->cdb;
 		if (cdb->byte2 & SVFY_BYTCHK)
 			bytchk = 1;
 		if (cdb->byte2 & SVFY_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_4btoul(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		break;
 	}
 	case VERIFY_16: {
 		struct scsi_rw_16 *cdb;
 
 		cdb = (struct scsi_rw_16 *)ctsio->cdb;
 		if (cdb->byte2 & SVFY_BYTCHK)
 			bytchk = 1;
 		if (cdb->byte2 & SVFY_DPO)
 			flags |= CTL_LLF_DPO;
 		lba = scsi_8btou64(cdb->addr);
 		num_blocks = scsi_4btoul(cdb->length);
 		break;
 	}
 	default:
 		/*
 		 * We got a command we don't support.  This shouldn't
 		 * happen, commands should be filtered out above us.
 		 */
 		ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * The first check is to make sure we're in bounds, the second
 	 * check is to catch wrap-around problems.  If the lba + num blocks
 	 * is less than the lba, then we've wrapped around and the block
 	 * range is invalid anyway.
 	 */
 	if (((lba + num_blocks) > (lun->be_lun->maxlba + 1))
 	 || ((lba + num_blocks) < lba)) {
 		ctl_set_lba_out_of_range(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	/*
 	 * According to SBC-3, a transfer length of 0 is not an error.
 	 */
 	if (num_blocks == 0) {
 		ctl_set_success(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	lbalen = (struct ctl_lba_len_flags *)
 	    &ctsio->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	lbalen->lba = lba;
 	lbalen->len = num_blocks;
 	if (bytchk) {
 		lbalen->flags = CTL_LLF_COMPARE | flags;
 		ctsio->kern_total_len = num_blocks * lun->be_lun->blocksize;
 	} else {
 		lbalen->flags = CTL_LLF_VERIFY | flags;
 		ctsio->kern_total_len = 0;
 	}
 	ctsio->kern_rel_offset = 0;
 
 	CTL_DEBUG_PRINT(("ctl_verify: calling data_submit()\n"));
 	retval = lun->backend->data_submit((union ctl_io *)ctsio);
 	return (retval);
 }
 
 int
 ctl_report_luns(struct ctl_scsiio *ctsio)
 {
 	struct ctl_softc *softc = control_softc;
 	struct scsi_report_luns *cdb;
 	struct scsi_report_luns_data *lun_data;
 	struct ctl_lun *lun, *request_lun;
 	int num_luns, retval;
 	uint32_t alloc_len, lun_datalen;
 	int num_filled, well_known;
 	uint32_t initidx, targ_lun_id, lun_id;
 
 	retval = CTL_RETVAL_COMPLETE;
 	well_known = 0;
 
 	cdb = (struct scsi_report_luns *)ctsio->cdb;
 
 	CTL_DEBUG_PRINT(("ctl_report_luns\n"));
 
 	mtx_lock(&softc->ctl_lock);
 	num_luns = softc->num_luns;
 	mtx_unlock(&softc->ctl_lock);
 
 	switch (cdb->select_report) {
 	case RPL_REPORT_DEFAULT:
 	case RPL_REPORT_ALL:
 		break;
 	case RPL_REPORT_WELLKNOWN:
 		well_known = 1;
 		num_luns = 0;
 		break;
 	default:
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (retval);
 		break; /* NOTREACHED */
 	}
 
 	alloc_len = scsi_4btoul(cdb->length);
 	/*
 	 * The initiator has to allocate at least 16 bytes for this request,
 	 * so he can at least get the header and the first LUN.  Otherwise
 	 * we reject the request (per SPC-3 rev 14, section 6.21).
 	 */
 	if (alloc_len < (sizeof(struct scsi_report_luns_data) +
 	    sizeof(struct scsi_report_luns_lundata))) {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 6,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (retval);
 	}
 
 	request_lun = (struct ctl_lun *)
 		ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	lun_datalen = sizeof(*lun_data) +
 		(num_luns * sizeof(struct scsi_report_luns_lundata));
 
 	ctsio->kern_data_ptr = malloc(lun_datalen, M_CTL, M_WAITOK | M_ZERO);
 	lun_data = (struct scsi_report_luns_data *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	initidx = ctl_get_initindex(&ctsio->io_hdr.nexus);
 
 	mtx_lock(&softc->ctl_lock);
 	for (targ_lun_id = 0, num_filled = 0; targ_lun_id < CTL_MAX_LUNS && num_filled < num_luns; targ_lun_id++) {
 		lun_id = ctl_map_lun(softc, ctsio->io_hdr.nexus.targ_port,
 		    targ_lun_id);
 		if (lun_id >= CTL_MAX_LUNS)
 			continue;
 		lun = softc->ctl_luns[lun_id];
 		if (lun == NULL)
 			continue;
 
 		if (targ_lun_id <= 0xff) {
 			/*
 			 * Peripheral addressing method, bus number 0.
 			 */
 			lun_data->luns[num_filled].lundata[0] =
 				RPL_LUNDATA_ATYP_PERIPH;
 			lun_data->luns[num_filled].lundata[1] = targ_lun_id;
 			num_filled++;
 		} else if (targ_lun_id <= 0x3fff) {
 			/*
 			 * Flat addressing method.
 			 */
 			lun_data->luns[num_filled].lundata[0] =
 				RPL_LUNDATA_ATYP_FLAT | (targ_lun_id >> 8);
 			lun_data->luns[num_filled].lundata[1] =
 				(targ_lun_id & 0xff);
 			num_filled++;
 		} else if (targ_lun_id <= 0xffffff) {
 			/*
 			 * Extended flat addressing method.
 			 */
 			lun_data->luns[num_filled].lundata[0] =
 			    RPL_LUNDATA_ATYP_EXTLUN | 0x12;
 			scsi_ulto3b(targ_lun_id,
 			    &lun_data->luns[num_filled].lundata[1]);
 			num_filled++;
 		} else {
 			printf("ctl_report_luns: bogus LUN number %jd, "
 			       "skipping\n", (intmax_t)targ_lun_id);
 		}
 		/*
 		 * According to SPC-3, rev 14 section 6.21:
 		 *
 		 * "The execution of a REPORT LUNS command to any valid and
 		 * installed logical unit shall clear the REPORTED LUNS DATA
 		 * HAS CHANGED unit attention condition for all logical
 		 * units of that target with respect to the requesting
 		 * initiator. A valid and installed logical unit is one
 		 * having a PERIPHERAL QUALIFIER of 000b in the standard
 		 * INQUIRY data (see 6.4.2)."
 		 *
 		 * If request_lun is NULL, the LUN this report luns command
 		 * was issued to is either disabled or doesn't exist. In that
 		 * case, we shouldn't clear any pending lun change unit
 		 * attention.
 		 */
 		if (request_lun != NULL) {
 			mtx_lock(&lun->lun_lock);
 			ctl_clr_ua(lun, initidx, CTL_UA_RES_RELEASE);
 			mtx_unlock(&lun->lun_lock);
 		}
 	}
 	mtx_unlock(&softc->ctl_lock);
 
 	/*
 	 * It's quite possible that we've returned fewer LUNs than we allocated
 	 * space for.  Trim it.
 	 */
 	lun_datalen = sizeof(*lun_data) +
 		(num_filled * sizeof(struct scsi_report_luns_lundata));
 
 	if (lun_datalen < alloc_len) {
 		ctsio->residual = alloc_len - lun_datalen;
 		ctsio->kern_data_len = lun_datalen;
 		ctsio->kern_total_len = lun_datalen;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * We set this to the actual data length, regardless of how much
 	 * space we actually have to return results.  If the user looks at
 	 * this value, he'll know whether or not he allocated enough space
 	 * and reissue the command if necessary.  We don't support well
 	 * known logical units, so if the user asks for that, return none.
 	 */
 	scsi_ulto4b(lun_datalen - 8, lun_data->length);
 
 	/*
 	 * We can only return SCSI_STATUS_CHECK_COND when we can't satisfy
 	 * this request.
 	 */
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (retval);
 }
 
 int
 ctl_request_sense(struct ctl_scsiio *ctsio)
 {
 	struct scsi_request_sense *cdb;
 	struct scsi_sense_data *sense_ptr;
+	struct ctl_softc *ctl_softc;
 	struct ctl_lun *lun;
 	uint32_t initidx;
 	int have_error;
 	scsi_sense_data_type sense_format;
 	ctl_ua_type ua_type;
 
 	cdb = (struct scsi_request_sense *)ctsio->cdb;
 
+	ctl_softc = control_softc;
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	CTL_DEBUG_PRINT(("ctl_request_sense\n"));
 
 	/*
 	 * Determine which sense format the user wants.
 	 */
 	if (cdb->byte2 & SRS_DESC)
 		sense_format = SSD_TYPE_DESC;
 	else
 		sense_format = SSD_TYPE_FIXED;
 
 	ctsio->kern_data_ptr = malloc(sizeof(*sense_ptr), M_CTL, M_WAITOK);
 	sense_ptr = (struct scsi_sense_data *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * struct scsi_sense_data, which is currently set to 256 bytes, is
 	 * larger than the largest allowed value for the length field in the
 	 * REQUEST SENSE CDB, which is 252 bytes as of SPC-4.
 	 */
 	ctsio->residual = 0;
 	ctsio->kern_data_len = cdb->length;
 	ctsio->kern_total_len = cdb->length;
 
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * If we don't have a LUN, we don't have any pending sense.
 	 */
 	if (lun == NULL)
 		goto no_sense;
 
 	have_error = 0;
 	initidx = ctl_get_initindex(&ctsio->io_hdr.nexus);
 	/*
 	 * Check for pending sense, and then for pending unit attentions.
 	 * Pending sense gets returned first, then pending unit attentions.
 	 */
 	mtx_lock(&lun->lun_lock);
 #ifdef CTL_WITH_CA
 	if (ctl_is_set(lun->have_ca, initidx)) {
 		scsi_sense_data_type stored_format;
 
 		/*
 		 * Check to see which sense format was used for the stored
 		 * sense data.
 		 */
 		stored_format = scsi_sense_type(&lun->pending_sense[initidx]);
 
 		/*
 		 * If the user requested a different sense format than the
 		 * one we stored, then we need to convert it to the other
 		 * format.  If we're going from descriptor to fixed format
 		 * sense data, we may lose things in translation, depending
 		 * on what options were used.
 		 *
 		 * If the stored format is SSD_TYPE_NONE (i.e. invalid),
 		 * for some reason we'll just copy it out as-is.
 		 */
 		if ((stored_format == SSD_TYPE_FIXED)
 		 && (sense_format == SSD_TYPE_DESC))
 			ctl_sense_to_desc((struct scsi_sense_data_fixed *)
 			    &lun->pending_sense[initidx],
 			    (struct scsi_sense_data_desc *)sense_ptr);
 		else if ((stored_format == SSD_TYPE_DESC)
 		      && (sense_format == SSD_TYPE_FIXED))
 			ctl_sense_to_fixed((struct scsi_sense_data_desc *)
 			    &lun->pending_sense[initidx],
 			    (struct scsi_sense_data_fixed *)sense_ptr);
 		else
 			memcpy(sense_ptr, &lun->pending_sense[initidx],
 			       MIN(sizeof(*sense_ptr),
 			       sizeof(lun->pending_sense[initidx])));
 
 		ctl_clear_mask(lun->have_ca, initidx);
 		have_error = 1;
 	} else
 #endif
 	{
 		ua_type = ctl_build_ua(lun, initidx, sense_ptr, sense_format);
 		if (ua_type != CTL_UA_NONE)
 			have_error = 1;
+		if (ua_type == CTL_UA_LUN_CHANGE) {
+			mtx_unlock(&lun->lun_lock);
+			mtx_lock(&ctl_softc->ctl_lock);
+			ctl_clear_ua(ctl_softc, initidx, ua_type);
+			mtx_unlock(&ctl_softc->ctl_lock);
+			mtx_lock(&lun->lun_lock);
+		}
+
 	}
 	mtx_unlock(&lun->lun_lock);
 
 	/*
 	 * We already have a pending error, return it.
 	 */
 	if (have_error != 0) {
 		/*
 		 * We report the SCSI status as OK, since the status of the
 		 * request sense command itself is OK.
 		 * We report 0 for the sense length, because we aren't doing
 		 * autosense in this case.  We're reporting sense as
 		 * parameter data.
 		 */
 		ctl_set_success(ctsio);
 		ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 		ctsio->be_move_done = ctl_config_move_done;
 		ctl_datamove((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 no_sense:
 
 	/*
 	 * No sense information to report, so we report that everything is
 	 * okay.
 	 */
 	ctl_set_sense_data(sense_ptr,
 			   lun,
 			   sense_format,
 			   /*current_error*/ 1,
 			   /*sense_key*/ SSD_KEY_NO_SENSE,
 			   /*asc*/ 0x00,
 			   /*ascq*/ 0x00,
 			   SSD_ELEM_NONE);
 
 	/*
 	 * We report 0 for the sense length, because we aren't doing
 	 * autosense in this case.  We're reporting sense as parameter data.
 	 */
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_tur(struct ctl_scsiio *ctsio)
 {
 
 	CTL_DEBUG_PRINT(("ctl_tur\n"));
 
 	ctl_set_success(ctsio);
 	ctl_done((union ctl_io *)ctsio);
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 #ifdef notyet
 static int
 ctl_cmddt_inquiry(struct ctl_scsiio *ctsio)
 {
 
 }
 #endif
 
+/*
+ * SCSI VPD page 0x00, the Supported VPD Pages page.
+ */
 static int
 ctl_inquiry_evpd_supported(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_supported_pages *pages;
 	int sup_page_size;
 	struct ctl_lun *lun;
 	int p;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	sup_page_size = sizeof(struct scsi_vpd_supported_pages) *
 	    SCSI_EVPD_NUM_SUPPORTED_PAGES;
 	ctsio->kern_data_ptr = malloc(sup_page_size, M_CTL, M_WAITOK | M_ZERO);
 	pages = (struct scsi_vpd_supported_pages *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (sup_page_size < alloc_len) {
 		ctsio->residual = alloc_len - sup_page_size;
 		ctsio->kern_data_len = sup_page_size;
 		ctsio->kern_total_len = sup_page_size;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.  Need to change this
 	 * to figure out whether the disk device is actually online or not.
 	 */
 	if (lun != NULL)
 		pages->device = (SID_QUAL_LU_CONNECTED << 5) |
 				lun->be_lun->lun_type;
 	else
 		pages->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 
 	p = 0;
 	/* Supported VPD pages */
 	pages->page_list[p++] = SVPD_SUPPORTED_PAGES;
 	/* Serial Number */
 	pages->page_list[p++] = SVPD_UNIT_SERIAL_NUMBER;
 	/* Device Identification */
 	pages->page_list[p++] = SVPD_DEVICE_ID;
 	/* Extended INQUIRY Data */
 	pages->page_list[p++] = SVPD_EXTENDED_INQUIRY_DATA;
 	/* Mode Page Policy */
 	pages->page_list[p++] = SVPD_MODE_PAGE_POLICY;
 	/* SCSI Ports */
 	pages->page_list[p++] = SVPD_SCSI_PORTS;
 	/* Third-party Copy */
 	pages->page_list[p++] = SVPD_SCSI_TPC;
 	if (lun != NULL && lun->be_lun->lun_type == T_DIRECT) {
 		/* Block limits */
 		pages->page_list[p++] = SVPD_BLOCK_LIMITS;
 		/* Block Device Characteristics */
 		pages->page_list[p++] = SVPD_BDC;
 		/* Logical Block Provisioning */
 		pages->page_list[p++] = SVPD_LBP;
 	}
 	pages->length = p;
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
+/*
+ * SCSI VPD page 0x80, the Unit Serial Number page.
+ */
 static int
 ctl_inquiry_evpd_serial(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_unit_serial_number *sn_ptr;
 	struct ctl_lun *lun;
 	int data_len;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	data_len = 4 + CTL_SN_LEN;
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	sn_ptr = (struct scsi_vpd_unit_serial_number *)ctsio->kern_data_ptr;
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.  Need to change this
 	 * to figure out whether the disk device is actually online or not.
 	 */
 	if (lun != NULL)
 		sn_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				  lun->be_lun->lun_type;
 	else
 		sn_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 
 	sn_ptr->page_code = SVPD_UNIT_SERIAL_NUMBER;
 	sn_ptr->length = CTL_SN_LEN;
 	/*
 	 * If we don't have a LUN, we just leave the serial number as
 	 * all spaces.
 	 */
 	if (lun != NULL) {
 		strncpy((char *)sn_ptr->serial_num,
 			(char *)lun->be_lun->serial_num, CTL_SN_LEN);
 	} else
 		memset(sn_ptr->serial_num, 0x20, CTL_SN_LEN);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 
+/*
+ * SCSI VPD page 0x86, the Extended INQUIRY Data page.
+ */
 static int
 ctl_inquiry_evpd_eid(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_extended_inquiry_data *eid_ptr;
 	struct ctl_lun *lun;
 	int data_len;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	data_len = sizeof(struct scsi_vpd_extended_inquiry_data);
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	eid_ptr = (struct scsi_vpd_extended_inquiry_data *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.
 	 */
 	if (lun != NULL)
 		eid_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				     lun->be_lun->lun_type;
 	else
 		eid_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 	eid_ptr->page_code = SVPD_EXTENDED_INQUIRY_DATA;
-	eid_ptr->page_length = data_len - 4;
+	scsi_ulto2b(data_len - 4, eid_ptr->page_length);
+	/*
+	 * We support head of queue, ordered and simple tags.
+	 */
 	eid_ptr->flags2 = SVPD_EID_HEADSUP | SVPD_EID_ORDSUP | SVPD_EID_SIMPSUP;
+	/*
+	 * Volatile cache supported.
+	 */
 	eid_ptr->flags3 = SVPD_EID_V_SUP;
 
+	/*
+	 * This means that we clear the REPORTED LUNS DATA HAS CHANGED unit
+	 * attention for a particular IT nexus on all LUNs once we report
+	 * it to that nexus once.  This bit is required as of SPC-4.
+	 */
+	eid_ptr->flags4 = SVPD_EID_LUICLT;
+
+	/*
+	 * XXX KDM in order to correctly answer this, we would need
+	 * information from the SIM to determine how much sense data it
+	 * can send.  So this would really be a path inquiry field, most
+	 * likely.  This can be set to a maximum of 252 according to SPC-4,
+	 * but the hardware may or may not be able to support that much.
+	 * 0 just means that the maximum sense data length is not reported.
+	 */
+	eid_ptr->max_sense_length = 0;
+
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 ctl_inquiry_evpd_mpp(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_mode_page_policy *mpp_ptr;
 	struct ctl_lun *lun;
 	int data_len;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	data_len = sizeof(struct scsi_vpd_mode_page_policy) +
 	    sizeof(struct scsi_vpd_mode_page_policy_descr);
 
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	mpp_ptr = (struct scsi_vpd_mode_page_policy *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.
 	 */
 	if (lun != NULL)
 		mpp_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				     lun->be_lun->lun_type;
 	else
 		mpp_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 	mpp_ptr->page_code = SVPD_MODE_PAGE_POLICY;
 	scsi_ulto2b(data_len - 4, mpp_ptr->page_length);
 	mpp_ptr->descr[0].page_code = 0x3f;
 	mpp_ptr->descr[0].subpage_code = 0xff;
 	mpp_ptr->descr[0].policy = SVPD_MPP_SHARED;
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
+/*
+ * SCSI VPD page 0x83, the Device Identification page.
+ */
 static int
 ctl_inquiry_evpd_devid(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_device_id *devid_ptr;
 	struct scsi_vpd_id_descriptor *desc;
 	struct ctl_softc *softc;
 	struct ctl_lun *lun;
 	struct ctl_port *port;
 	int data_len;
 	uint8_t proto;
 
 	softc = control_softc;
 
 	port = softc->ctl_ports[ctl_port_idx(ctsio->io_hdr.nexus.targ_port)];
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	data_len = sizeof(struct scsi_vpd_device_id) +
 	    sizeof(struct scsi_vpd_id_descriptor) +
 		sizeof(struct scsi_vpd_id_rel_trgt_port_id) +
 	    sizeof(struct scsi_vpd_id_descriptor) +
 		sizeof(struct scsi_vpd_id_trgt_port_grp_id);
 	if (lun && lun->lun_devid)
 		data_len += lun->lun_devid->len;
 	if (port->port_devid)
 		data_len += port->port_devid->len;
 	if (port->target_devid)
 		data_len += port->target_devid->len;
 
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	devid_ptr = (struct scsi_vpd_device_id *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.
 	 */
 	if (lun != NULL)
 		devid_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				     lun->be_lun->lun_type;
 	else
 		devid_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 	devid_ptr->page_code = SVPD_DEVICE_ID;
 	scsi_ulto2b(data_len - 4, devid_ptr->length);
 
 	if (port->port_type == CTL_PORT_FC)
 		proto = SCSI_PROTO_FC << 4;
 	else if (port->port_type == CTL_PORT_ISCSI)
 		proto = SCSI_PROTO_ISCSI << 4;
 	else
 		proto = SCSI_PROTO_SPI << 4;
 	desc = (struct scsi_vpd_id_descriptor *)devid_ptr->desc_list;
 
 	/*
 	 * We're using a LUN association here.  i.e., this device ID is a
 	 * per-LUN identifier.
 	 */
 	if (lun && lun->lun_devid) {
 		memcpy(desc, lun->lun_devid->data, lun->lun_devid->len);
 		desc = (struct scsi_vpd_id_descriptor *)((uint8_t *)desc +
 		    lun->lun_devid->len);
 	}
 
 	/*
 	 * This is for the WWPN which is a port association.
 	 */
 	if (port->port_devid) {
 		memcpy(desc, port->port_devid->data, port->port_devid->len);
 		desc = (struct scsi_vpd_id_descriptor *)((uint8_t *)desc +
 		    port->port_devid->len);
 	}
 
 	/*
 	 * This is for the Relative Target Port(type 4h) identifier
 	 */
 	desc->proto_codeset = proto | SVPD_ID_CODESET_BINARY;
 	desc->id_type = SVPD_ID_PIV | SVPD_ID_ASSOC_PORT |
 	    SVPD_ID_TYPE_RELTARG;
 	desc->length = 4;
 	scsi_ulto2b(ctsio->io_hdr.nexus.targ_port, &desc->identifier[2]);
 	desc = (struct scsi_vpd_id_descriptor *)(&desc->identifier[0] +
 	    sizeof(struct scsi_vpd_id_rel_trgt_port_id));
 
 	/*
 	 * This is for the Target Port Group(type 5h) identifier
 	 */
 	desc->proto_codeset = proto | SVPD_ID_CODESET_BINARY;
 	desc->id_type = SVPD_ID_PIV | SVPD_ID_ASSOC_PORT |
 	    SVPD_ID_TYPE_TPORTGRP;
 	desc->length = 4;
 	scsi_ulto2b(ctsio->io_hdr.nexus.targ_port / CTL_MAX_PORTS + 1,
 	    &desc->identifier[2]);
 	desc = (struct scsi_vpd_id_descriptor *)(&desc->identifier[0] +
 	    sizeof(struct scsi_vpd_id_trgt_port_grp_id));
 
 	/*
 	 * This is for the Target identifier
 	 */
 	if (port->target_devid) {
 		memcpy(desc, port->target_devid->data, port->target_devid->len);
 	}
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 ctl_inquiry_evpd_scsi_ports(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct ctl_softc *softc = control_softc;
 	struct scsi_vpd_scsi_ports *sp;
 	struct scsi_vpd_port_designation *pd;
 	struct scsi_vpd_port_designation_cont *pdc;
 	struct ctl_lun *lun;
 	struct ctl_port *port;
 	int data_len, num_target_ports, iid_len, id_len, g, pg, p;
 	int num_target_port_groups;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	if (softc->is_single)
 		num_target_port_groups = 1;
 	else
 		num_target_port_groups = NUM_TARGET_PORT_GROUPS;
 	num_target_ports = 0;
 	iid_len = 0;
 	id_len = 0;
 	mtx_lock(&softc->ctl_lock);
 	STAILQ_FOREACH(port, &softc->port_list, links) {
 		if ((port->status & CTL_PORT_STATUS_ONLINE) == 0)
 			continue;
 		if (lun != NULL &&
 		    ctl_map_lun_back(softc, port->targ_port, lun->lun) >=
 		    CTL_MAX_LUNS)
 			continue;
 		num_target_ports++;
 		if (port->init_devid)
 			iid_len += port->init_devid->len;
 		if (port->port_devid)
 			id_len += port->port_devid->len;
 	}
 	mtx_unlock(&softc->ctl_lock);
 
 	data_len = sizeof(struct scsi_vpd_scsi_ports) + num_target_port_groups *
 	    num_target_ports * (sizeof(struct scsi_vpd_port_designation) +
 	     sizeof(struct scsi_vpd_port_designation_cont)) + iid_len + id_len;
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	sp = (struct scsi_vpd_scsi_ports *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.  Need to change this
 	 * to figure out whether the disk device is actually online or not.
 	 */
 	if (lun != NULL)
 		sp->device = (SID_QUAL_LU_CONNECTED << 5) |
 				  lun->be_lun->lun_type;
 	else
 		sp->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 
 	sp->page_code = SVPD_SCSI_PORTS;
 	scsi_ulto2b(data_len - sizeof(struct scsi_vpd_scsi_ports),
 	    sp->page_length);
 	pd = &sp->design[0];
 
 	mtx_lock(&softc->ctl_lock);
 	pg = softc->port_offset / CTL_MAX_PORTS;
 	for (g = 0; g < num_target_port_groups; g++) {
 		STAILQ_FOREACH(port, &softc->port_list, links) {
 			if ((port->status & CTL_PORT_STATUS_ONLINE) == 0)
 				continue;
 			if (lun != NULL &&
 			    ctl_map_lun_back(softc, port->targ_port, lun->lun)
 			    >= CTL_MAX_LUNS)
 				continue;
 			p = port->targ_port % CTL_MAX_PORTS + g * CTL_MAX_PORTS;
 			scsi_ulto2b(p, pd->relative_port_id);
 			if (port->init_devid && g == pg) {
 				iid_len = port->init_devid->len;
 				memcpy(pd->initiator_transportid,
 				    port->init_devid->data, port->init_devid->len);
 			} else
 				iid_len = 0;
 			scsi_ulto2b(iid_len, pd->initiator_transportid_length);
 			pdc = (struct scsi_vpd_port_designation_cont *)
 			    (&pd->initiator_transportid[iid_len]);
 			if (port->port_devid && g == pg) {
 				id_len = port->port_devid->len;
 				memcpy(pdc->target_port_descriptors,
 				    port->port_devid->data, port->port_devid->len);
 			} else
 				id_len = 0;
 			scsi_ulto2b(id_len, pdc->target_port_descriptors_length);
 			pd = (struct scsi_vpd_port_designation *)
 			    ((uint8_t *)pdc->target_port_descriptors + id_len);
 		}
 	}
 	mtx_unlock(&softc->ctl_lock);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 ctl_inquiry_evpd_block_limits(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_block_limits *bl_ptr;
 	struct ctl_lun *lun;
 	int bs;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	ctsio->kern_data_ptr = malloc(sizeof(*bl_ptr), M_CTL, M_WAITOK | M_ZERO);
 	bl_ptr = (struct scsi_vpd_block_limits *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (sizeof(*bl_ptr) < alloc_len) {
 		ctsio->residual = alloc_len - sizeof(*bl_ptr);
 		ctsio->kern_data_len = sizeof(*bl_ptr);
 		ctsio->kern_total_len = sizeof(*bl_ptr);
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.  Need to change this
 	 * to figure out whether the disk device is actually online or not.
 	 */
 	if (lun != NULL)
 		bl_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				  lun->be_lun->lun_type;
 	else
 		bl_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 
 	bl_ptr->page_code = SVPD_BLOCK_LIMITS;
 	scsi_ulto2b(sizeof(*bl_ptr) - 4, bl_ptr->page_length);
 	bl_ptr->max_cmp_write_len = 0xff;
 	scsi_ulto4b(0xffffffff, bl_ptr->max_txfer_len);
 	if (lun != NULL) {
 		bs = lun->be_lun->blocksize;
 		scsi_ulto4b(lun->be_lun->opttxferlen, bl_ptr->opt_txfer_len);
 		if (lun->be_lun->flags & CTL_LUN_FLAG_UNMAP) {
 			scsi_ulto4b(0xffffffff, bl_ptr->max_unmap_lba_cnt);
 			scsi_ulto4b(0xffffffff, bl_ptr->max_unmap_blk_cnt);
 			if (lun->be_lun->ublockexp != 0) {
 				scsi_ulto4b((1 << lun->be_lun->ublockexp),
 				    bl_ptr->opt_unmap_grain);
 				scsi_ulto4b(0x80000000 | lun->be_lun->ublockoff,
 				    bl_ptr->unmap_grain_align);
 			}
 		}
 		scsi_ulto4b(lun->be_lun->atomicblock,
 		    bl_ptr->max_atomic_transfer_length);
 		scsi_ulto4b(0, bl_ptr->atomic_alignment);
 		scsi_ulto4b(0, bl_ptr->atomic_transfer_length_granularity);
 	}
 	scsi_u64to8b(UINT64_MAX, bl_ptr->max_write_same_length);
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 ctl_inquiry_evpd_bdc(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_block_device_characteristics *bdc_ptr;
 	struct ctl_lun *lun;
 	const char *value;
 	u_int i;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	ctsio->kern_data_ptr = malloc(sizeof(*bdc_ptr), M_CTL, M_WAITOK | M_ZERO);
 	bdc_ptr = (struct scsi_vpd_block_device_characteristics *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (sizeof(*bdc_ptr) < alloc_len) {
 		ctsio->residual = alloc_len - sizeof(*bdc_ptr);
 		ctsio->kern_data_len = sizeof(*bdc_ptr);
 		ctsio->kern_total_len = sizeof(*bdc_ptr);
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.  Need to change this
 	 * to figure out whether the disk device is actually online or not.
 	 */
 	if (lun != NULL)
 		bdc_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				  lun->be_lun->lun_type;
 	else
 		bdc_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 	bdc_ptr->page_code = SVPD_BDC;
 	scsi_ulto2b(sizeof(*bdc_ptr) - 4, bdc_ptr->page_length);
 	if (lun != NULL &&
 	    (value = ctl_get_opt(&lun->be_lun->options, "rpm")) != NULL)
 		i = strtol(value, NULL, 0);
 	else
 		i = CTL_DEFAULT_ROTATION_RATE;
 	scsi_ulto2b(i, bdc_ptr->medium_rotation_rate);
 	if (lun != NULL &&
 	    (value = ctl_get_opt(&lun->be_lun->options, "formfactor")) != NULL)
 		i = strtol(value, NULL, 0);
 	else
 		i = 0;
 	bdc_ptr->wab_wac_ff = (i & 0x0f);
 	bdc_ptr->flags = SVPD_FUAB | SVPD_VBULS;
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 static int
 ctl_inquiry_evpd_lbp(struct ctl_scsiio *ctsio, int alloc_len)
 {
 	struct scsi_vpd_logical_block_prov *lbp_ptr;
 	struct ctl_lun *lun;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	ctsio->kern_data_ptr = malloc(sizeof(*lbp_ptr), M_CTL, M_WAITOK | M_ZERO);
 	lbp_ptr = (struct scsi_vpd_logical_block_prov *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 
 	if (sizeof(*lbp_ptr) < alloc_len) {
 		ctsio->residual = alloc_len - sizeof(*lbp_ptr);
 		ctsio->kern_data_len = sizeof(*lbp_ptr);
 		ctsio->kern_total_len = sizeof(*lbp_ptr);
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 	ctsio->kern_sg_entries = 0;
 
 	/*
 	 * The control device is always connected.  The disk device, on the
 	 * other hand, may not be online all the time.  Need to change this
 	 * to figure out whether the disk device is actually online or not.
 	 */
 	if (lun != NULL)
 		lbp_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				  lun->be_lun->lun_type;
 	else
 		lbp_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 
 	lbp_ptr->page_code = SVPD_LBP;
 	scsi_ulto2b(sizeof(*lbp_ptr) - 4, lbp_ptr->page_length);
 	lbp_ptr->threshold_exponent = CTL_LBP_EXPONENT;
 	if (lun != NULL && lun->be_lun->flags & CTL_LUN_FLAG_UNMAP) {
 		lbp_ptr->flags = SVPD_LBP_UNMAP | SVPD_LBP_WS16 |
 		    SVPD_LBP_WS10 | SVPD_LBP_RZ | SVPD_LBP_ANC_SUP;
 		lbp_ptr->prov_type = SVPD_LBP_THIN;
 	}
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
+/*
+ * INQUIRY with the EVPD bit set.
+ */
 static int
 ctl_inquiry_evpd(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	struct scsi_inquiry *cdb;
 	int alloc_len, retval;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_inquiry *)ctsio->cdb;
 	alloc_len = scsi_2btoul(cdb->length);
 
 	switch (cdb->page_code) {
 	case SVPD_SUPPORTED_PAGES:
 		retval = ctl_inquiry_evpd_supported(ctsio, alloc_len);
 		break;
 	case SVPD_UNIT_SERIAL_NUMBER:
 		retval = ctl_inquiry_evpd_serial(ctsio, alloc_len);
 		break;
 	case SVPD_DEVICE_ID:
 		retval = ctl_inquiry_evpd_devid(ctsio, alloc_len);
 		break;
 	case SVPD_EXTENDED_INQUIRY_DATA:
 		retval = ctl_inquiry_evpd_eid(ctsio, alloc_len);
 		break;
 	case SVPD_MODE_PAGE_POLICY:
 		retval = ctl_inquiry_evpd_mpp(ctsio, alloc_len);
 		break;
 	case SVPD_SCSI_PORTS:
 		retval = ctl_inquiry_evpd_scsi_ports(ctsio, alloc_len);
 		break;
 	case SVPD_SCSI_TPC:
 		retval = ctl_inquiry_evpd_tpc(ctsio, alloc_len);
 		break;
 	case SVPD_BLOCK_LIMITS:
 		if (lun == NULL || lun->be_lun->lun_type != T_DIRECT)
 			goto err;
 		retval = ctl_inquiry_evpd_block_limits(ctsio, alloc_len);
 		break;
 	case SVPD_BDC:
 		if (lun == NULL || lun->be_lun->lun_type != T_DIRECT)
 			goto err;
 		retval = ctl_inquiry_evpd_bdc(ctsio, alloc_len);
 		break;
 	case SVPD_LBP:
 		if (lun == NULL || lun->be_lun->lun_type != T_DIRECT)
 			goto err;
 		retval = ctl_inquiry_evpd_lbp(ctsio, alloc_len);
 		break;
 	default:
 err:
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		retval = CTL_RETVAL_COMPLETE;
 		break;
 	}
 
 	return (retval);
 }
 
+/*
+ * Standard INQUIRY data.
+ */
 static int
 ctl_inquiry_std(struct ctl_scsiio *ctsio)
 {
 	struct scsi_inquiry_data *inq_ptr;
 	struct scsi_inquiry *cdb;
 	struct ctl_softc *softc;
 	struct ctl_lun *lun;
 	char *val;
 	uint32_t alloc_len, data_len;
 	ctl_port_type port_type;
 
 	softc = control_softc;
 
 	/*
 	 * Figure out whether we're talking to a Fibre Channel port or not.
 	 * We treat the ioctl front end, and any SCSI adapters, as packetized
 	 * SCSI front ends.
 	 */
 	port_type = softc->ctl_ports[
 	    ctl_port_idx(ctsio->io_hdr.nexus.targ_port)]->port_type;
 	if (port_type == CTL_PORT_IOCTL || port_type == CTL_PORT_INTERNAL)
 		port_type = CTL_PORT_SCSI;
 
 	lun = ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	cdb = (struct scsi_inquiry *)ctsio->cdb;
 	alloc_len = scsi_2btoul(cdb->length);
 
 	/*
 	 * We malloc the full inquiry data size here and fill it
 	 * in.  If the user only asks for less, we'll give him
 	 * that much.
 	 */
 	data_len = offsetof(struct scsi_inquiry_data, vendor_specific1);
 	ctsio->kern_data_ptr = malloc(data_len, M_CTL, M_WAITOK | M_ZERO);
 	inq_ptr = (struct scsi_inquiry_data *)ctsio->kern_data_ptr;
 	ctsio->kern_sg_entries = 0;
 	ctsio->kern_data_resid = 0;
 	ctsio->kern_rel_offset = 0;
 
 	if (data_len < alloc_len) {
 		ctsio->residual = alloc_len - data_len;
 		ctsio->kern_data_len = data_len;
 		ctsio->kern_total_len = data_len;
 	} else {
 		ctsio->residual = 0;
 		ctsio->kern_data_len = alloc_len;
 		ctsio->kern_total_len = alloc_len;
 	}
 
 	/*
 	 * If we have a LUN configured, report it as connected.  Otherwise,
 	 * report that it is offline or no device is supported, depending 
 	 * on the value of inquiry_pq_no_lun.
 	 *
 	 * According to the spec (SPC-4 r34), the peripheral qualifier
 	 * SID_QUAL_LU_OFFLINE (001b) is used in the following scenario:
 	 *
 	 * "A peripheral device having the specified peripheral device type 
 	 * is not connected to this logical unit. However, the device
 	 * server is capable of supporting the specified peripheral device
 	 * type on this logical unit."
 	 *
 	 * According to the same spec, the peripheral qualifier
 	 * SID_QUAL_BAD_LU (011b) is used in this scenario:
 	 *
 	 * "The device server is not capable of supporting a peripheral
 	 * device on this logical unit. For this peripheral qualifier the
 	 * peripheral device type shall be set to 1Fh. All other peripheral
 	 * device type values are reserved for this peripheral qualifier."
 	 *
 	 * Given the text, it would seem that we probably want to report that
 	 * the LUN is offline here.  There is no LUN connected, but we can
 	 * support a LUN at the given LUN number.
 	 *
 	 * In the real world, though, it sounds like things are a little
 	 * different:
 	 *
 	 * - Linux, when presented with a LUN with the offline peripheral
 	 *   qualifier, will create an sg driver instance for it.  So when
 	 *   you attach it to CTL, you wind up with a ton of sg driver
 	 *   instances.  (One for every LUN that Linux bothered to probe.)
 	 *   Linux does this despite the fact that it issues a REPORT LUNs
 	 *   to LUN 0 to get the inventory of supported LUNs.
 	 *
 	 * - There is other anecdotal evidence (from Emulex folks) about
 	 *   arrays that use the offline peripheral qualifier for LUNs that
 	 *   are on the "passive" path in an active/passive array.
 	 *
 	 * So the solution is provide a hopefully reasonable default
 	 * (return bad/no LUN) and allow the user to change the behavior
 	 * with a tunable/sysctl variable.
 	 */
 	if (lun != NULL)
 		inq_ptr->device = (SID_QUAL_LU_CONNECTED << 5) |
 				  lun->be_lun->lun_type;
 	else if (softc->inquiry_pq_no_lun == 0)
 		inq_ptr->device = (SID_QUAL_LU_OFFLINE << 5) | T_DIRECT;
 	else
 		inq_ptr->device = (SID_QUAL_BAD_LU << 5) | T_NODEVICE;
 
 	/* RMB in byte 2 is 0 */
 	inq_ptr->version = SCSI_REV_SPC4;
 
 	/*
 	 * According to SAM-3, even if a device only supports a single
 	 * level of LUN addressing, it should still set the HISUP bit:
 	 *
 	 * 4.9.1 Logical unit numbers overview
 	 *
 	 * All logical unit number formats described in this standard are
 	 * hierarchical in structure even when only a single level in that
 	 * hierarchy is used. The HISUP bit shall be set to one in the
 	 * standard INQUIRY data (see SPC-2) when any logical unit number
 	 * format described in this standard is used.  Non-hierarchical
 	 * formats are outside the scope of this standard.
 	 *
 	 * Therefore we set the HiSup bit here.
 	 *
 	 * The reponse format is 2, per SPC-3.
 	 */
 	inq_ptr->response_format = SID_HiSup | 2;
 
 	inq_ptr->additional_length = data_len -
 	    (offsetof(struct scsi_inquiry_data, additional_length) + 1);
 	CTL_DEBUG_PRINT(("additional_length = %d\n",
 			 inq_ptr->additional_length));
 
 	inq_ptr->spc3_flags = SPC3_SID_3PC | SPC3_SID_TPGS_IMPLICIT;
 	/* 16 bit addressing */
 	if (port_type == CTL_PORT_SCSI)
 		inq_ptr->spc2_flags = SPC2_SID_ADDR16;
 	/* XXX set the SID_MultiP bit here if we're actually going to
 	   respond on multiple ports */
 	inq_ptr->spc2_flags |= SPC2_SID_MultiP;
 
 	/* 16 bit data bus, synchronous transfers */
 	if (port_type == CTL_PORT_SCSI)
 		inq_ptr->flags = SID_WBus16 | SID_Sync;
 	/*
 	 * XXX KDM do we want to support tagged queueing on the control
 	 * device at all?
 	 */
 	if ((lun == NULL)
 	 || (lun->be_lun->lun_type != T_PROCESSOR))
 		inq_ptr->flags |= SID_CmdQue;
 	/*
 	 * Per SPC-3, unused bytes in ASCII strings are filled with spaces.
 	 * We have 8 bytes for the vendor name, and 16 bytes for the device
 	 * name and 4 bytes for the revision.
 	 */
 	if (lun == NULL || (val = ctl_get_opt(&lun->be_lun->options,
 	    "vendor")) == NULL) {
 		strncpy(inq_ptr->vendor, CTL_VENDOR, sizeof(inq_ptr->vendor));
 	} else {
 		memset(inq_ptr->vendor, ' ', sizeof(inq_ptr->vendor));
 		strncpy(inq_ptr->vendor, val,
 		    min(sizeof(inq_ptr->vendor), strlen(val)));
 	}
 	if (lun == NULL) {
 		strncpy(inq_ptr->product, CTL_DIRECT_PRODUCT,
 		    sizeof(inq_ptr->product));
 	} else if ((val = ctl_get_opt(&lun->be_lun->options, "product")) == NULL) {
 		switch (lun->be_lun->lun_type) {
 		case T_DIRECT:
 			strncpy(inq_ptr->product, CTL_DIRECT_PRODUCT,
 			    sizeof(inq_ptr->product));
 			break;
 		case T_PROCESSOR:
 			strncpy(inq_ptr->product, CTL_PROCESSOR_PRODUCT,
 			    sizeof(inq_ptr->product));
 			break;
 		default:
 			strncpy(inq_ptr->product, CTL_UNKNOWN_PRODUCT,
 			    sizeof(inq_ptr->product));
 			break;
 		}
 	} else {
 		memset(inq_ptr->product, ' ', sizeof(inq_ptr->product));
 		strncpy(inq_ptr->product, val,
 		    min(sizeof(inq_ptr->product), strlen(val)));
 	}
 
 	/*
 	 * XXX make this a macro somewhere so it automatically gets
 	 * incremented when we make changes.
 	 */
 	if (lun == NULL || (val = ctl_get_opt(&lun->be_lun->options,
 	    "revision")) == NULL) {
 		strncpy(inq_ptr->revision, "0001", sizeof(inq_ptr->revision));
 	} else {
 		memset(inq_ptr->revision, ' ', sizeof(inq_ptr->revision));
 		strncpy(inq_ptr->revision, val,
 		    min(sizeof(inq_ptr->revision), strlen(val)));
 	}
 
 	/*
 	 * For parallel SCSI, we support double transition and single
 	 * transition clocking.  We also support QAS (Quick Arbitration
 	 * and Selection) and Information Unit transfers on both the
 	 * control and array devices.
 	 */
 	if (port_type == CTL_PORT_SCSI)
 		inq_ptr->spi3data = SID_SPI_CLOCK_DT_ST | SID_SPI_QAS |
 				    SID_SPI_IUS;
 
 	/* SAM-5 (no version claimed) */
 	scsi_ulto2b(0x00A0, inq_ptr->version1);
 	/* SPC-4 (no version claimed) */
 	scsi_ulto2b(0x0460, inq_ptr->version2);
 	if (port_type == CTL_PORT_FC) {
 		/* FCP-2 ANSI INCITS.350:2003 */
 		scsi_ulto2b(0x0917, inq_ptr->version3);
 	} else if (port_type == CTL_PORT_SCSI) {
 		/* SPI-4 ANSI INCITS.362:200x */
 		scsi_ulto2b(0x0B56, inq_ptr->version3);
 	} else if (port_type == CTL_PORT_ISCSI) {
 		/* iSCSI (no version claimed) */
 		scsi_ulto2b(0x0960, inq_ptr->version3);
 	} else if (port_type == CTL_PORT_SAS) {
 		/* SAS (no version claimed) */
 		scsi_ulto2b(0x0BE0, inq_ptr->version3);
 	}
 
 	if (lun == NULL) {
 		/* SBC-4 (no version claimed) */
 		scsi_ulto2b(0x0600, inq_ptr->version4);
 	} else {
 		switch (lun->be_lun->lun_type) {
 		case T_DIRECT:
 			/* SBC-4 (no version claimed) */
 			scsi_ulto2b(0x0600, inq_ptr->version4);
 			break;
 		case T_PROCESSOR:
 		default:
 			break;
 		}
 	}
 
 	ctl_set_success(ctsio);
 	ctsio->io_hdr.flags |= CTL_FLAG_ALLOCATED;
 	ctsio->be_move_done = ctl_config_move_done;
 	ctl_datamove((union ctl_io *)ctsio);
 	return (CTL_RETVAL_COMPLETE);
 }
 
 int
 ctl_inquiry(struct ctl_scsiio *ctsio)
 {
 	struct scsi_inquiry *cdb;
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_inquiry\n"));
 
 	cdb = (struct scsi_inquiry *)ctsio->cdb;
 	if (cdb->byte2 & SI_EVPD)
 		retval = ctl_inquiry_evpd(ctsio);
 	else if (cdb->page_code == 0)
 		retval = ctl_inquiry_std(ctsio);
 	else {
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ 2,
 				      /*bit_valid*/ 0,
 				      /*bit*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		return (CTL_RETVAL_COMPLETE);
 	}
 
 	return (retval);
 }
 
 /*
  * For known CDB types, parse the LBA and length.
  */
 static int
 ctl_get_lba_len(union ctl_io *io, uint64_t *lba, uint64_t *len)
 {
 	if (io->io_hdr.io_type != CTL_IO_SCSI)
 		return (1);
 
 	switch (io->scsiio.cdb[0]) {
 	case COMPARE_AND_WRITE: {
 		struct scsi_compare_and_write *cdb;
 
 		cdb = (struct scsi_compare_and_write *)io->scsiio.cdb;
 
 		*lba = scsi_8btou64(cdb->addr);
 		*len = cdb->length;
 		break;
 	}
 	case READ_6:
 	case WRITE_6: {
 		struct scsi_rw_6 *cdb;
 
 		cdb = (struct scsi_rw_6 *)io->scsiio.cdb;
 
 		*lba = scsi_3btoul(cdb->addr);
 		/* only 5 bits are valid in the most significant address byte */
 		*lba &= 0x1fffff;
 		*len = cdb->length;
 		break;
 	}
 	case READ_10:
 	case WRITE_10: {
 		struct scsi_rw_10 *cdb;
 
 		cdb = (struct scsi_rw_10 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_2btoul(cdb->length);
 		break;
 	}
 	case WRITE_VERIFY_10: {
 		struct scsi_write_verify_10 *cdb;
 
 		cdb = (struct scsi_write_verify_10 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_2btoul(cdb->length);
 		break;
 	}
 	case READ_12:
 	case WRITE_12: {
 		struct scsi_rw_12 *cdb;
 
 		cdb = (struct scsi_rw_12 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case WRITE_VERIFY_12: {
 		struct scsi_write_verify_12 *cdb;
 
 		cdb = (struct scsi_write_verify_12 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case READ_16:
 	case WRITE_16:
 	case WRITE_ATOMIC_16: {
 		struct scsi_rw_16 *cdb;
 
 		cdb = (struct scsi_rw_16 *)io->scsiio.cdb;
 
 		*lba = scsi_8btou64(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case WRITE_VERIFY_16: {
 		struct scsi_write_verify_16 *cdb;
 
 		cdb = (struct scsi_write_verify_16 *)io->scsiio.cdb;
 
 		*lba = scsi_8btou64(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case WRITE_SAME_10: {
 		struct scsi_write_same_10 *cdb;
 
 		cdb = (struct scsi_write_same_10 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_2btoul(cdb->length);
 		break;
 	}
 	case WRITE_SAME_16: {
 		struct scsi_write_same_16 *cdb;
 
 		cdb = (struct scsi_write_same_16 *)io->scsiio.cdb;
 
 		*lba = scsi_8btou64(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case VERIFY_10: {
 		struct scsi_verify_10 *cdb;
 
 		cdb = (struct scsi_verify_10 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_2btoul(cdb->length);
 		break;
 	}
 	case VERIFY_12: {
 		struct scsi_verify_12 *cdb;
 
 		cdb = (struct scsi_verify_12 *)io->scsiio.cdb;
 
 		*lba = scsi_4btoul(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case VERIFY_16: {
 		struct scsi_verify_16 *cdb;
 
 		cdb = (struct scsi_verify_16 *)io->scsiio.cdb;
 
 		*lba = scsi_8btou64(cdb->addr);
 		*len = scsi_4btoul(cdb->length);
 		break;
 	}
 	case UNMAP: {
 		*lba = 0;
 		*len = UINT64_MAX;
 		break;
 	}
 	case SERVICE_ACTION_IN: {	/* GET LBA STATUS */
 		struct scsi_get_lba_status *cdb;
 
 		cdb = (struct scsi_get_lba_status *)io->scsiio.cdb;
 		*lba = scsi_8btou64(cdb->addr);
 		*len = UINT32_MAX;
 		break;
 	}
 	default:
 		return (1);
 		break; /* NOTREACHED */
 	}
 
 	return (0);
 }
 
 static ctl_action
 ctl_extent_check_lba(uint64_t lba1, uint64_t len1, uint64_t lba2, uint64_t len2,
     bool seq)
 {
 	uint64_t endlba1, endlba2;
 
 	endlba1 = lba1 + len1 - (seq ? 0 : 1);
 	endlba2 = lba2 + len2 - 1;
 
 	if ((endlba1 < lba2) || (endlba2 < lba1))
 		return (CTL_ACTION_PASS);
 	else
 		return (CTL_ACTION_BLOCK);
 }
 
 static int
 ctl_extent_check_unmap(union ctl_io *io, uint64_t lba2, uint64_t len2)
 {
 	struct ctl_ptr_len_flags *ptrlen;
 	struct scsi_unmap_desc *buf, *end, *range;
 	uint64_t lba;
 	uint32_t len;
 
 	/* If not UNMAP -- go other way. */
 	if (io->io_hdr.io_type != CTL_IO_SCSI ||
 	    io->scsiio.cdb[0] != UNMAP)
 		return (CTL_ACTION_ERROR);
 
 	/* If UNMAP without data -- block and wait for data. */
 	ptrlen = (struct ctl_ptr_len_flags *)
 	    &io->io_hdr.ctl_private[CTL_PRIV_LBA_LEN];
 	if ((io->io_hdr.flags & CTL_FLAG_ALLOCATED) == 0 ||
 	    ptrlen->ptr == NULL)
 		return (CTL_ACTION_BLOCK);
 
 	/* UNMAP with data -- check for collision. */
 	buf = (struct scsi_unmap_desc *)ptrlen->ptr;
 	end = buf + ptrlen->len / sizeof(*buf);
 	for (range = buf; range < end; range++) {
 		lba = scsi_8btou64(range->lba);
 		len = scsi_4btoul(range->length);
 		if ((lba < lba2 + len2) && (lba + len > lba2))
 			return (CTL_ACTION_BLOCK);
 	}
 	return (CTL_ACTION_PASS);
 }
 
 static ctl_action
 ctl_extent_check(union ctl_io *io1, union ctl_io *io2, bool seq)
 {
 	uint64_t lba1, lba2;
 	uint64_t len1, len2;
 	int retval;
 
 	if (ctl_get_lba_len(io2, &lba2, &len2) != 0)
 		return (CTL_ACTION_ERROR);
 
 	retval = ctl_extent_check_unmap(io1, lba2, len2);
 	if (retval != CTL_ACTION_ERROR)
 		return (retval);
 
 	if (ctl_get_lba_len(io1, &lba1, &len1) != 0)
 		return (CTL_ACTION_ERROR);
 
 	return (ctl_extent_check_lba(lba1, len1, lba2, len2, seq));
 }
 
 static ctl_action
 ctl_extent_check_seq(union ctl_io *io1, union ctl_io *io2)
 {
 	uint64_t lba1, lba2;
 	uint64_t len1, len2;
 
 	if (ctl_get_lba_len(io1, &lba1, &len1) != 0)
 		return (CTL_ACTION_ERROR);
 	if (ctl_get_lba_len(io2, &lba2, &len2) != 0)
 		return (CTL_ACTION_ERROR);
 
 	if (lba1 + len1 == lba2)
 		return (CTL_ACTION_BLOCK);
 	return (CTL_ACTION_PASS);
 }
 
 static ctl_action
 ctl_check_for_blockage(struct ctl_lun *lun, union ctl_io *pending_io,
     union ctl_io *ooa_io)
 {
 	const struct ctl_cmd_entry *pending_entry, *ooa_entry;
 	ctl_serialize_action *serialize_row;
 
 	/*
 	 * The initiator attempted multiple untagged commands at the same
 	 * time.  Can't do that.
 	 */
 	if ((pending_io->scsiio.tag_type == CTL_TAG_UNTAGGED)
 	 && (ooa_io->scsiio.tag_type == CTL_TAG_UNTAGGED)
 	 && ((pending_io->io_hdr.nexus.targ_port ==
 	      ooa_io->io_hdr.nexus.targ_port)
 	  && (pending_io->io_hdr.nexus.initid.id ==
 	      ooa_io->io_hdr.nexus.initid.id))
 	 && ((ooa_io->io_hdr.flags & (CTL_FLAG_ABORT |
 	      CTL_FLAG_STATUS_SENT)) == 0))
 		return (CTL_ACTION_OVERLAP);
 
 	/*
 	 * The initiator attempted to send multiple tagged commands with
 	 * the same ID.  (It's fine if different initiators have the same
 	 * tag ID.)
 	 *
 	 * Even if all of those conditions are true, we don't kill the I/O
 	 * if the command ahead of us has been aborted.  We won't end up
 	 * sending it to the FETD, and it's perfectly legal to resend a
 	 * command with the same tag number as long as the previous
 	 * instance of this tag number has been aborted somehow.
 	 */
 	if ((pending_io->scsiio.tag_type != CTL_TAG_UNTAGGED)
 	 && (ooa_io->scsiio.tag_type != CTL_TAG_UNTAGGED)
 	 && (pending_io->scsiio.tag_num == ooa_io->scsiio.tag_num)
 	 && ((pending_io->io_hdr.nexus.targ_port ==
 	      ooa_io->io_hdr.nexus.targ_port)
 	  && (pending_io->io_hdr.nexus.initid.id ==
 	      ooa_io->io_hdr.nexus.initid.id))
 	 && ((ooa_io->io_hdr.flags & (CTL_FLAG_ABORT |
 	      CTL_FLAG_STATUS_SENT)) == 0))
 		return (CTL_ACTION_OVERLAP_TAG);
 
 	/*
 	 * If we get a head of queue tag, SAM-3 says that we should
 	 * immediately execute it.
 	 *
 	 * What happens if this command would normally block for some other
 	 * reason?  e.g. a request sense with a head of queue tag
 	 * immediately after a write.  Normally that would block, but this
 	 * will result in its getting executed immediately...
 	 *
 	 * We currently return "pass" instead of "skip", so we'll end up
 	 * going through the rest of the queue to check for overlapped tags.
 	 *
 	 * XXX KDM check for other types of blockage first??
 	 */
 	if (pending_io->scsiio.tag_type == CTL_TAG_HEAD_OF_QUEUE)
 		return (CTL_ACTION_PASS);
 
 	/*
 	 * Ordered tags have to block until all items ahead of them
 	 * have completed.  If we get called with an ordered tag, we always
 	 * block, if something else is ahead of us in the queue.
 	 */
 	if (pending_io->scsiio.tag_type == CTL_TAG_ORDERED)
 		return (CTL_ACTION_BLOCK);
 
 	/*
 	 * Simple tags get blocked until all head of queue and ordered tags
 	 * ahead of them have completed.  I'm lumping untagged commands in
 	 * with simple tags here.  XXX KDM is that the right thing to do?
 	 */
 	if (((pending_io->scsiio.tag_type == CTL_TAG_UNTAGGED)
 	  || (pending_io->scsiio.tag_type == CTL_TAG_SIMPLE))
 	 && ((ooa_io->scsiio.tag_type == CTL_TAG_HEAD_OF_QUEUE)
 	  || (ooa_io->scsiio.tag_type == CTL_TAG_ORDERED)))
 		return (CTL_ACTION_BLOCK);
 
 	pending_entry = ctl_get_cmd_entry(&pending_io->scsiio, NULL);
 	ooa_entry = ctl_get_cmd_entry(&ooa_io->scsiio, NULL);
 
 	serialize_row = ctl_serialize_table[ooa_entry->seridx];
 
 	switch (serialize_row[pending_entry->seridx]) {
 	case CTL_SER_BLOCK:
 		return (CTL_ACTION_BLOCK);
 	case CTL_SER_EXTENT:
 		return (ctl_extent_check(ooa_io, pending_io,
 		    (lun->serseq == CTL_LUN_SERSEQ_ON)));
 	case CTL_SER_EXTENTOPT:
 		if ((lun->mode_pages.control_page[CTL_PAGE_CURRENT].queue_flags
 		    & SCP_QUEUE_ALG_MASK) != SCP_QUEUE_ALG_UNRESTRICTED)
 			return (ctl_extent_check(ooa_io, pending_io,
 			    (lun->serseq == CTL_LUN_SERSEQ_ON)));
 		return (CTL_ACTION_PASS);
 	case CTL_SER_EXTENTSEQ:
 		if (lun->serseq != CTL_LUN_SERSEQ_OFF)
 			return (ctl_extent_check_seq(ooa_io, pending_io));
 		return (CTL_ACTION_PASS);
 	case CTL_SER_PASS:
 		return (CTL_ACTION_PASS);
 	case CTL_SER_BLOCKOPT:
 		if ((lun->mode_pages.control_page[CTL_PAGE_CURRENT].queue_flags
 		    & SCP_QUEUE_ALG_MASK) != SCP_QUEUE_ALG_UNRESTRICTED)
 			return (CTL_ACTION_BLOCK);
 		return (CTL_ACTION_PASS);
 	case CTL_SER_SKIP:
 		return (CTL_ACTION_SKIP);
 	default:
 		panic("invalid serialization value %d",
 		      serialize_row[pending_entry->seridx]);
 	}
 
 	return (CTL_ACTION_ERROR);
 }
 
 /*
  * Check for blockage or overlaps against the OOA (Order Of Arrival) queue.
  * Assumptions:
  * - pending_io is generally either incoming, or on the blocked queue
  * - starting I/O is the I/O we want to start the check with.
  */
 static ctl_action
 ctl_check_ooa(struct ctl_lun *lun, union ctl_io *pending_io,
 	      union ctl_io *starting_io)
 {
 	union ctl_io *ooa_io;
 	ctl_action action;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 
 	/*
 	 * Run back along the OOA queue, starting with the current
 	 * blocked I/O and going through every I/O before it on the
 	 * queue.  If starting_io is NULL, we'll just end up returning
 	 * CTL_ACTION_PASS.
 	 */
 	for (ooa_io = starting_io; ooa_io != NULL;
 	     ooa_io = (union ctl_io *)TAILQ_PREV(&ooa_io->io_hdr, ctl_ooaq,
 	     ooa_links)){
 
 		/*
 		 * This routine just checks to see whether
 		 * cur_blocked is blocked by ooa_io, which is ahead
 		 * of it in the queue.  It doesn't queue/dequeue
 		 * cur_blocked.
 		 */
 		action = ctl_check_for_blockage(lun, pending_io, ooa_io);
 		switch (action) {
 		case CTL_ACTION_BLOCK:
 		case CTL_ACTION_OVERLAP:
 		case CTL_ACTION_OVERLAP_TAG:
 		case CTL_ACTION_SKIP:
 		case CTL_ACTION_ERROR:
 			return (action);
 			break; /* NOTREACHED */
 		case CTL_ACTION_PASS:
 			break;
 		default:
 			panic("invalid action %d", action);
 			break;  /* NOTREACHED */
 		}
 	}
 
 	return (CTL_ACTION_PASS);
 }
 
 /*
  * Assumptions:
  * - An I/O has just completed, and has been removed from the per-LUN OOA
  *   queue, so some items on the blocked queue may now be unblocked.
  */
 static int
 ctl_check_blocked(struct ctl_lun *lun)
 {
 	union ctl_io *cur_blocked, *next_blocked;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 
 	/*
 	 * Run forward from the head of the blocked queue, checking each
 	 * entry against the I/Os prior to it on the OOA queue to see if
 	 * there is still any blockage.
 	 *
 	 * We cannot use the TAILQ_FOREACH() macro, because it can't deal
 	 * with our removing a variable on it while it is traversing the
 	 * list.
 	 */
 	for (cur_blocked = (union ctl_io *)TAILQ_FIRST(&lun->blocked_queue);
 	     cur_blocked != NULL; cur_blocked = next_blocked) {
 		union ctl_io *prev_ooa;
 		ctl_action action;
 
 		next_blocked = (union ctl_io *)TAILQ_NEXT(&cur_blocked->io_hdr,
 							  blocked_links);
 
 		prev_ooa = (union ctl_io *)TAILQ_PREV(&cur_blocked->io_hdr,
 						      ctl_ooaq, ooa_links);
 
 		/*
 		 * If cur_blocked happens to be the first item in the OOA
 		 * queue now, prev_ooa will be NULL, and the action
 		 * returned will just be CTL_ACTION_PASS.
 		 */
 		action = ctl_check_ooa(lun, cur_blocked, prev_ooa);
 
 		switch (action) {
 		case CTL_ACTION_BLOCK:
 			/* Nothing to do here, still blocked */
 			break;
 		case CTL_ACTION_OVERLAP:
 		case CTL_ACTION_OVERLAP_TAG:
 			/*
 			 * This shouldn't happen!  In theory we've already
 			 * checked this command for overlap...
 			 */
 			break;
 		case CTL_ACTION_PASS:
 		case CTL_ACTION_SKIP: {
 			const struct ctl_cmd_entry *entry;
 			int isc_retval;
 
 			/*
 			 * The skip case shouldn't happen, this transaction
 			 * should have never made it onto the blocked queue.
 			 */
 			/*
 			 * This I/O is no longer blocked, we can remove it
 			 * from the blocked queue.  Since this is a TAILQ
 			 * (doubly linked list), we can do O(1) removals
 			 * from any place on the list.
 			 */
 			TAILQ_REMOVE(&lun->blocked_queue, &cur_blocked->io_hdr,
 				     blocked_links);
 			cur_blocked->io_hdr.flags &= ~CTL_FLAG_BLOCKED;
 
 			if (cur_blocked->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC){
 				/*
 				 * Need to send IO back to original side to
 				 * run
 				 */
 				union ctl_ha_msg msg_info;
 
 				msg_info.hdr.original_sc =
 					cur_blocked->io_hdr.original_sc;
 				msg_info.hdr.serializing_sc = cur_blocked;
 				msg_info.hdr.msg_type = CTL_MSG_R2R;
 				if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 				     &msg_info, sizeof(msg_info), 0)) >
 				     CTL_HA_STATUS_SUCCESS) {
 					printf("CTL:Check Blocked error from "
 					       "ctl_ha_msg_send %d\n",
 					       isc_retval);
 				}
 				break;
 			}
 			entry = ctl_get_cmd_entry(&cur_blocked->scsiio, NULL);
 
 			/*
 			 * Check this I/O for LUN state changes that may
 			 * have happened while this command was blocked.
 			 * The LUN state may have been changed by a command
 			 * ahead of us in the queue, so we need to re-check
 			 * for any states that can be caused by SCSI
 			 * commands.
 			 */
 			if (ctl_scsiio_lun_check(lun, entry,
 						 &cur_blocked->scsiio) == 0) {
 				cur_blocked->io_hdr.flags |=
 				                      CTL_FLAG_IS_WAS_ON_RTR;
 				ctl_enqueue_rtr(cur_blocked);
 			} else
 				ctl_done(cur_blocked);
 			break;
 		}
 		default:
 			/*
 			 * This probably shouldn't happen -- we shouldn't
 			 * get CTL_ACTION_ERROR, or anything else.
 			 */
 			break;
 		}
 	}
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 /*
  * This routine (with one exception) checks LUN flags that can be set by
  * commands ahead of us in the OOA queue.  These flags have to be checked
  * when a command initially comes in, and when we pull a command off the
  * blocked queue and are preparing to execute it.  The reason we have to
  * check these flags for commands on the blocked queue is that the LUN
  * state may have been changed by a command ahead of us while we're on the
  * blocked queue.
  *
  * Ordering is somewhat important with these checks, so please pay
  * careful attention to the placement of any new checks.
  */
 static int
 ctl_scsiio_lun_check(struct ctl_lun *lun,
     const struct ctl_cmd_entry *entry, struct ctl_scsiio *ctsio)
 {
 	struct ctl_softc *softc = lun->ctl_softc;
 	int retval;
 	uint32_t residx;
 
 	retval = 0;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 
 	/*
 	 * If this shelf is a secondary shelf controller, we have to reject
 	 * any media access commands.
 	 */
 	if ((softc->flags & CTL_FLAG_ACTIVE_SHELF) == 0 &&
 	    (entry->flags & CTL_CMD_FLAG_OK_ON_SECONDARY) == 0) {
 		ctl_set_lun_standby(ctsio);
 		retval = 1;
 		goto bailout;
 	}
 
 	if (entry->pattern & CTL_LUN_PAT_WRITE) {
 		if (lun->flags & CTL_LUN_READONLY) {
 			ctl_set_sense(ctsio, /*current_error*/ 1,
 			    /*sense_key*/ SSD_KEY_DATA_PROTECT,
 			    /*asc*/ 0x27, /*ascq*/ 0x01, SSD_ELEM_NONE);
 			retval = 1;
 			goto bailout;
 		}
 		if ((lun->mode_pages.control_page[CTL_PAGE_CURRENT]
 		    .eca_and_aen & SCP_SWP) != 0) {
 			ctl_set_sense(ctsio, /*current_error*/ 1,
 			    /*sense_key*/ SSD_KEY_DATA_PROTECT,
 			    /*asc*/ 0x27, /*ascq*/ 0x02, SSD_ELEM_NONE);
 			retval = 1;
 			goto bailout;
 		}
 	}
 
 	/*
 	 * Check for a reservation conflict.  If this command isn't allowed
 	 * even on reserved LUNs, and if this initiator isn't the one who
 	 * reserved us, reject the command with a reservation conflict.
 	 */
 	residx = ctl_get_resindex(&ctsio->io_hdr.nexus);
 	if ((lun->flags & CTL_LUN_RESERVED)
 	 && ((entry->flags & CTL_CMD_FLAG_ALLOW_ON_RESV) == 0)) {
 		if (lun->res_idx != residx) {
 			ctl_set_reservation_conflict(ctsio);
 			retval = 1;
 			goto bailout;
 		}
 	}
 
 	if ((lun->flags & CTL_LUN_PR_RESERVED) == 0 ||
 	    (entry->flags & CTL_CMD_FLAG_ALLOW_ON_PR_RESV)) {
 		/* No reservation or command is allowed. */;
 	} else if ((entry->flags & CTL_CMD_FLAG_ALLOW_ON_PR_WRESV) &&
 	    (lun->res_type == SPR_TYPE_WR_EX ||
 	     lun->res_type == SPR_TYPE_WR_EX_RO ||
 	     lun->res_type == SPR_TYPE_WR_EX_AR)) {
 		/* The command is allowed for Write Exclusive resv. */;
 	} else {
 		/*
 		 * if we aren't registered or it's a res holder type
 		 * reservation and this isn't the res holder then set a
 		 * conflict.
 		 */
 		if (ctl_get_prkey(lun, residx) == 0
 		 || (residx != lun->pr_res_idx && lun->res_type < 4)) {
 			ctl_set_reservation_conflict(ctsio);
 			retval = 1;
 			goto bailout;
 		}
 
 	}
 
 	if ((lun->flags & CTL_LUN_OFFLINE)
 	 && ((entry->flags & CTL_CMD_FLAG_OK_ON_OFFLINE) == 0)) {
 		ctl_set_lun_not_ready(ctsio);
 		retval = 1;
 		goto bailout;
 	}
 
 	/*
 	 * If the LUN is stopped, see if this particular command is allowed
 	 * for a stopped lun.  Otherwise, reject it with 0x04,0x02.
 	 */
 	if ((lun->flags & CTL_LUN_STOPPED)
 	 && ((entry->flags & CTL_CMD_FLAG_OK_ON_STOPPED) == 0)) {
 		/* "Logical unit not ready, initializing cmd. required" */
 		ctl_set_lun_stopped(ctsio);
 		retval = 1;
 		goto bailout;
 	}
 
 	if ((lun->flags & CTL_LUN_INOPERABLE)
 	 && ((entry->flags & CTL_CMD_FLAG_OK_ON_INOPERABLE) == 0)) {
 		/* "Medium format corrupted" */
 		ctl_set_medium_format_corrupted(ctsio);
 		retval = 1;
 		goto bailout;
 	}
 
 bailout:
 	return (retval);
 
 }
 
 static void
 ctl_failover_io(union ctl_io *io, int have_lock)
 {
 	ctl_set_busy(&io->scsiio);
 	ctl_done(io);
 }
 
 static void
 ctl_failover(void)
 {
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	union ctl_io *next_io, *pending_io;
 	union ctl_io *io;
 	int lun_idx;
 
 	softc = control_softc;
 
 	mtx_lock(&softc->ctl_lock);
 	/*
 	 * Remove any cmds from the other SC from the rtr queue.  These
 	 * will obviously only be for LUNs for which we're the primary.
 	 * We can't send status or get/send data for these commands.
 	 * Since they haven't been executed yet, we can just remove them.
 	 * We'll either abort them or delete them below, depending on
 	 * which HA mode we're in.
 	 */
 #ifdef notyet
 	mtx_lock(&softc->queue_lock);
 	for (io = (union ctl_io *)STAILQ_FIRST(&softc->rtr_queue);
 	     io != NULL; io = next_io) {
 		next_io = (union ctl_io *)STAILQ_NEXT(&io->io_hdr, links);
 		if (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC)
 			STAILQ_REMOVE(&softc->rtr_queue, &io->io_hdr,
 				      ctl_io_hdr, links);
 	}
 	mtx_unlock(&softc->queue_lock);
 #endif
 
 	for (lun_idx=0; lun_idx < softc->num_luns; lun_idx++) {
 		lun = softc->ctl_luns[lun_idx];
 		if (lun==NULL)
 			continue;
 
 		/*
 		 * Processor LUNs are primary on both sides.
 		 * XXX will this always be true?
 		 */
 		if (lun->be_lun->lun_type == T_PROCESSOR)
 			continue;
 
 		if ((lun->flags & CTL_LUN_PRIMARY_SC)
 		 && (softc->ha_mode == CTL_HA_MODE_SER_ONLY)) {
 			printf("FAILOVER: primary lun %d\n", lun_idx);
 		        /*
 			 * Remove all commands from the other SC. First from the
 			 * blocked queue then from the ooa queue. Once we have
 			 * removed them. Call ctl_check_blocked to see if there
 			 * is anything that can run.
 			 */
 			for (io = (union ctl_io *)TAILQ_FIRST(
 			     &lun->blocked_queue); io != NULL; io = next_io) {
 
 		        	next_io = (union ctl_io *)TAILQ_NEXT(
 				    &io->io_hdr, blocked_links);
 
 				if (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC) {
 					TAILQ_REMOVE(&lun->blocked_queue,
 						     &io->io_hdr,blocked_links);
 					io->io_hdr.flags &= ~CTL_FLAG_BLOCKED;
 					TAILQ_REMOVE(&lun->ooa_queue,
 						     &io->io_hdr, ooa_links);
 
 					ctl_free_io(io);
 				}
 			}
 
 			for (io = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue);
 	     		     io != NULL; io = next_io) {
 
 		        	next_io = (union ctl_io *)TAILQ_NEXT(
 				    &io->io_hdr, ooa_links);
 
 				if (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC) {
 
 					TAILQ_REMOVE(&lun->ooa_queue,
 						&io->io_hdr,
 					     	ooa_links);
 
 					ctl_free_io(io);
 				}
 			}
 			ctl_check_blocked(lun);
 		} else if ((lun->flags & CTL_LUN_PRIMARY_SC)
 			&& (softc->ha_mode == CTL_HA_MODE_XFER)) {
 
 			printf("FAILOVER: primary lun %d\n", lun_idx);
 			/*
 			 * Abort all commands from the other SC.  We can't
 			 * send status back for them now.  These should get
 			 * cleaned up when they are completed or come out
 			 * for a datamove operation.
 			 */
 			for (io = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue);
 	     		     io != NULL; io = next_io) {
 		        	next_io = (union ctl_io *)TAILQ_NEXT(
 					&io->io_hdr, ooa_links);
 
 				if (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC)
 					io->io_hdr.flags |= CTL_FLAG_ABORT;
 			}
 		} else if (((lun->flags & CTL_LUN_PRIMARY_SC) == 0)
 			&& (softc->ha_mode == CTL_HA_MODE_XFER)) {
 
 			printf("FAILOVER: secondary lun %d\n", lun_idx);
 
 			lun->flags |= CTL_LUN_PRIMARY_SC;
 
 			/*
 			 * We send all I/O that was sent to this controller
 			 * and redirected to the other side back with
 			 * busy status, and have the initiator retry it.
 			 * Figuring out how much data has been transferred,
 			 * etc. and picking up where we left off would be 
 			 * very tricky.
 			 *
 			 * XXX KDM need to remove I/O from the blocked
 			 * queue as well!
 			 */
 			for (pending_io = (union ctl_io *)TAILQ_FIRST(
 			     &lun->ooa_queue); pending_io != NULL;
 			     pending_io = next_io) {
 
 				next_io =  (union ctl_io *)TAILQ_NEXT(
 					&pending_io->io_hdr, ooa_links);
 
 				pending_io->io_hdr.flags &=
 					~CTL_FLAG_SENT_2OTHER_SC;
 
 				if (pending_io->io_hdr.flags &
 				    CTL_FLAG_IO_ACTIVE) {
 					pending_io->io_hdr.flags |=
 						CTL_FLAG_FAILOVER;
 				} else {
 					ctl_set_busy(&pending_io->scsiio);
 					ctl_done(pending_io);
 				}
 			}
 
 			ctl_est_ua_all(lun, -1, CTL_UA_ASYM_ACC_CHANGE);
 		} else if (((lun->flags & CTL_LUN_PRIMARY_SC) == 0)
 			&& (softc->ha_mode == CTL_HA_MODE_SER_ONLY)) {
 			printf("FAILOVER: secondary lun %d\n", lun_idx);
 			/*
 			 * if the first io on the OOA is not on the RtR queue
 			 * add it.
 			 */
 			lun->flags |= CTL_LUN_PRIMARY_SC;
 
 			pending_io = (union ctl_io *)TAILQ_FIRST(
 			    &lun->ooa_queue);
 			if (pending_io==NULL) {
 				printf("Nothing on OOA queue\n");
 				continue;
 			}
 
 			pending_io->io_hdr.flags &= ~CTL_FLAG_SENT_2OTHER_SC;
 			if ((pending_io->io_hdr.flags &
 			     CTL_FLAG_IS_WAS_ON_RTR) == 0) {
 				pending_io->io_hdr.flags |=
 				    CTL_FLAG_IS_WAS_ON_RTR;
 				ctl_enqueue_rtr(pending_io);
 			}
 #if 0
 			else
 			{
 				printf("Tag 0x%04x is running\n",
 				      pending_io->scsiio.tag_num);
 			}
 #endif
 
 			next_io = (union ctl_io *)TAILQ_NEXT(
 			    &pending_io->io_hdr, ooa_links);
 			for (pending_io=next_io; pending_io != NULL;
 			     pending_io = next_io) {
 				pending_io->io_hdr.flags &=
 				    ~CTL_FLAG_SENT_2OTHER_SC;
 				next_io = (union ctl_io *)TAILQ_NEXT(
 					&pending_io->io_hdr, ooa_links);
 				if (pending_io->io_hdr.flags &
 				    CTL_FLAG_IS_WAS_ON_RTR) {
 #if 0
 				        printf("Tag 0x%04x is running\n",
 				      		pending_io->scsiio.tag_num);
 #endif
 					continue;
 				}
 
 				switch (ctl_check_ooa(lun, pending_io,
 			            (union ctl_io *)TAILQ_PREV(
 				    &pending_io->io_hdr, ctl_ooaq,
 				    ooa_links))) {
 
 				case CTL_ACTION_BLOCK:
 					TAILQ_INSERT_TAIL(&lun->blocked_queue,
 							  &pending_io->io_hdr,
 							  blocked_links);
 					pending_io->io_hdr.flags |=
 					    CTL_FLAG_BLOCKED;
 					break;
 				case CTL_ACTION_PASS:
 				case CTL_ACTION_SKIP:
 					pending_io->io_hdr.flags |=
 					    CTL_FLAG_IS_WAS_ON_RTR;
 					ctl_enqueue_rtr(pending_io);
 					break;
 				case CTL_ACTION_OVERLAP:
 					ctl_set_overlapped_cmd(
 					    (struct ctl_scsiio *)pending_io);
 					ctl_done(pending_io);
 					break;
 				case CTL_ACTION_OVERLAP_TAG:
 					ctl_set_overlapped_tag(
 					    (struct ctl_scsiio *)pending_io,
 					    pending_io->scsiio.tag_num & 0xff);
 					ctl_done(pending_io);
 					break;
 				case CTL_ACTION_ERROR:
 				default:
 					ctl_set_internal_failure(
 						(struct ctl_scsiio *)pending_io,
 						0,  // sks_valid
 						0); //retry count
 					ctl_done(pending_io);
 					break;
 				}
 			}
 
 			ctl_est_ua_all(lun, -1, CTL_UA_ASYM_ACC_CHANGE);
 		} else {
 			panic("Unhandled HA mode failover, LUN flags = %#x, "
 			      "ha_mode = #%x", lun->flags, softc->ha_mode);
 		}
 	}
 	ctl_pause_rtr = 0;
 	mtx_unlock(&softc->ctl_lock);
+}
+
+static void
+ctl_clear_ua(struct ctl_softc *ctl_softc, uint32_t initidx,
+	     ctl_ua_type ua_type)
+{
+	struct ctl_lun *lun;
+	ctl_ua_type *pu;
+
+	mtx_assert(&ctl_softc->ctl_lock, MA_OWNED);
+
+	STAILQ_FOREACH(lun, &ctl_softc->lun_list, links) {
+		mtx_lock(&lun->lun_lock);
+		pu = lun->pending_ua[initidx / CTL_MAX_INIT_PER_PORT];
+		pu[initidx % CTL_MAX_INIT_PER_PORT] &= ~ua_type;
+		mtx_unlock(&lun->lun_lock);
+	}
 }
 
 static int
 ctl_scsiio_precheck(struct ctl_softc *softc, struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	const struct ctl_cmd_entry *entry;
 	uint32_t initidx, targ_lun;
 	int retval;
 
 	retval = 0;
 
 	lun = NULL;
 
 	targ_lun = ctsio->io_hdr.nexus.targ_mapped_lun;
 	if ((targ_lun < CTL_MAX_LUNS)
 	 && ((lun = softc->ctl_luns[targ_lun]) != NULL)) {
 		/*
 		 * If the LUN is invalid, pretend that it doesn't exist.
 		 * It will go away as soon as all pending I/O has been
 		 * completed.
 		 */
 		mtx_lock(&lun->lun_lock);
 		if (lun->flags & CTL_LUN_DISABLED) {
 			mtx_unlock(&lun->lun_lock);
 			lun = NULL;
 			ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr = NULL;
 			ctsio->io_hdr.ctl_private[CTL_PRIV_BACKEND_LUN].ptr = NULL;
 		} else {
 			ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr = lun;
 			ctsio->io_hdr.ctl_private[CTL_PRIV_BACKEND_LUN].ptr =
 				lun->be_lun;
 			if (lun->be_lun->lun_type == T_PROCESSOR) {
 				ctsio->io_hdr.flags |= CTL_FLAG_CONTROL_DEV;
 			}
 
 			/*
 			 * Every I/O goes into the OOA queue for a
 			 * particular LUN, and stays there until completion.
 			 */
 			TAILQ_INSERT_TAIL(&lun->ooa_queue, &ctsio->io_hdr,
 			    ooa_links);
 		}
 	} else {
 		ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr = NULL;
 		ctsio->io_hdr.ctl_private[CTL_PRIV_BACKEND_LUN].ptr = NULL;
 	}
 
 	/* Get command entry and return error if it is unsuppotyed. */
 	entry = ctl_validate_command(ctsio);
 	if (entry == NULL) {
 		if (lun)
 			mtx_unlock(&lun->lun_lock);
 		return (retval);
 	}
 
 	ctsio->io_hdr.flags &= ~CTL_FLAG_DATA_MASK;
 	ctsio->io_hdr.flags |= entry->flags & CTL_FLAG_DATA_MASK;
 
 	/*
 	 * Check to see whether we can send this command to LUNs that don't
 	 * exist.  This should pretty much only be the case for inquiry
 	 * and request sense.  Further checks, below, really require having
 	 * a LUN, so we can't really check the command anymore.  Just put
 	 * it on the rtr queue.
 	 */
 	if (lun == NULL) {
 		if (entry->flags & CTL_CMD_FLAG_OK_ON_ALL_LUNS) {
 			ctsio->io_hdr.flags |= CTL_FLAG_IS_WAS_ON_RTR;
 			ctl_enqueue_rtr((union ctl_io *)ctsio);
 			return (retval);
 		}
 
 		ctl_set_unsupported_lun(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		CTL_DEBUG_PRINT(("ctl_scsiio_precheck: bailing out due to invalid LUN\n"));
 		return (retval);
 	} else {
 		/*
 		 * Make sure we support this particular command on this LUN.
 		 * e.g., we don't support writes to the control LUN.
 		 */
 		if (!ctl_cmd_applicable(lun->be_lun->lun_type, entry)) {
 			mtx_unlock(&lun->lun_lock);
 			ctl_set_invalid_opcode(ctsio);
 			ctl_done((union ctl_io *)ctsio);
 			return (retval);
 		}
 	}
 
 	initidx = ctl_get_initindex(&ctsio->io_hdr.nexus);
 
 #ifdef CTL_WITH_CA
 	/*
 	 * If we've got a request sense, it'll clear the contingent
 	 * allegiance condition.  Otherwise, if we have a CA condition for
 	 * this initiator, clear it, because it sent down a command other
 	 * than request sense.
 	 */
 	if ((ctsio->cdb[0] != REQUEST_SENSE)
 	 && (ctl_is_set(lun->have_ca, initidx)))
 		ctl_clear_mask(lun->have_ca, initidx);
 #endif
 
 	/*
 	 * If the command has this flag set, it handles its own unit
 	 * attention reporting, we shouldn't do anything.  Otherwise we
 	 * check for any pending unit attentions, and send them back to the
 	 * initiator.  We only do this when a command initially comes in,
 	 * not when we pull it off the blocked queue.
 	 *
 	 * According to SAM-3, section 5.3.2, the order that things get
 	 * presented back to the host is basically unit attentions caused
 	 * by some sort of reset event, busy status, reservation conflicts
 	 * or task set full, and finally any other status.
 	 *
 	 * One issue here is that some of the unit attentions we report
 	 * don't fall into the "reset" category (e.g. "reported luns data
 	 * has changed").  So reporting it here, before the reservation
 	 * check, may be technically wrong.  I guess the only thing to do
 	 * would be to check for and report the reset events here, and then
 	 * check for the other unit attention types after we check for a
 	 * reservation conflict.
 	 *
 	 * XXX KDM need to fix this
 	 */
 	if ((entry->flags & CTL_CMD_FLAG_NO_SENSE) == 0) {
 		ctl_ua_type ua_type;
 		scsi_sense_data_type sense_format;
 
 		if (lun->flags & CTL_LUN_SENSE_DESC)
 			sense_format = SSD_TYPE_DESC;
 		else
 			sense_format = SSD_TYPE_FIXED;
 
 		ua_type = ctl_build_ua(lun, initidx, &ctsio->sense_data,
 		    sense_format);
 		if (ua_type != CTL_UA_NONE) {
 			mtx_unlock(&lun->lun_lock);
 			ctsio->scsi_status = SCSI_STATUS_CHECK_COND;
 			ctsio->io_hdr.status = CTL_SCSI_ERROR | CTL_AUTOSENSE;
 			ctsio->sense_len = SSD_FULL_SIZE;
 			ctl_done((union ctl_io *)ctsio);
 			return (retval);
 		}
 	}
 
 
 	if (ctl_scsiio_lun_check(lun, entry, ctsio) != 0) {
 		mtx_unlock(&lun->lun_lock);
 		ctl_done((union ctl_io *)ctsio);
 		return (retval);
 	}
 
 	/*
 	 * XXX CHD this is where we want to send IO to other side if
 	 * this LUN is secondary on this SC. We will need to make a copy
 	 * of the IO and flag the IO on this side as SENT_2OTHER and the flag
 	 * the copy we send as FROM_OTHER.
 	 * We also need to stuff the address of the original IO so we can
 	 * find it easily. Something similar will need be done on the other
 	 * side so when we are done we can find the copy.
 	 */
 	if ((lun->flags & CTL_LUN_PRIMARY_SC) == 0) {
 		union ctl_ha_msg msg_info;
 		int isc_retval;
 
 		ctsio->io_hdr.flags |= CTL_FLAG_SENT_2OTHER_SC;
 
 		msg_info.hdr.msg_type = CTL_MSG_SERIALIZE;
 		msg_info.hdr.original_sc = (union ctl_io *)ctsio;
 #if 0
 		printf("1. ctsio %p\n", ctsio);
 #endif
 		msg_info.hdr.serializing_sc = NULL;
 		msg_info.hdr.nexus = ctsio->io_hdr.nexus;
 		msg_info.scsi.tag_num = ctsio->tag_num;
 		msg_info.scsi.tag_type = ctsio->tag_type;
 		memcpy(msg_info.scsi.cdb, ctsio->cdb, CTL_MAX_CDBLEN);
 
 		ctsio->io_hdr.flags &= ~CTL_FLAG_IO_ACTIVE;
 
 		if ((isc_retval=ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 		    (void *)&msg_info, sizeof(msg_info), 0)) >
 		    CTL_HA_STATUS_SUCCESS) {
 			printf("CTL:precheck, ctl_ha_msg_send returned %d\n",
 			       isc_retval);
 			printf("CTL:opcode is %x\n", ctsio->cdb[0]);
 		} else {
 #if 0
 			printf("CTL:Precheck sent msg, opcode is %x\n",opcode);
 #endif
 		}
 
 		/*
 		 * XXX KDM this I/O is off the incoming queue, but hasn't
 		 * been inserted on any other queue.  We may need to come
 		 * up with a holding queue while we wait for serialization
 		 * so that we have an idea of what we're waiting for from
 		 * the other side.
 		 */
 		mtx_unlock(&lun->lun_lock);
 		return (retval);
 	}
 
 	switch (ctl_check_ooa(lun, (union ctl_io *)ctsio,
 			      (union ctl_io *)TAILQ_PREV(&ctsio->io_hdr,
 			      ctl_ooaq, ooa_links))) {
 	case CTL_ACTION_BLOCK:
 		ctsio->io_hdr.flags |= CTL_FLAG_BLOCKED;
 		TAILQ_INSERT_TAIL(&lun->blocked_queue, &ctsio->io_hdr,
 				  blocked_links);
 		mtx_unlock(&lun->lun_lock);
 		return (retval);
 	case CTL_ACTION_PASS:
 	case CTL_ACTION_SKIP:
 		ctsio->io_hdr.flags |= CTL_FLAG_IS_WAS_ON_RTR;
 		mtx_unlock(&lun->lun_lock);
 		ctl_enqueue_rtr((union ctl_io *)ctsio);
 		break;
 	case CTL_ACTION_OVERLAP:
 		mtx_unlock(&lun->lun_lock);
 		ctl_set_overlapped_cmd(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		break;
 	case CTL_ACTION_OVERLAP_TAG:
 		mtx_unlock(&lun->lun_lock);
 		ctl_set_overlapped_tag(ctsio, ctsio->tag_num & 0xff);
 		ctl_done((union ctl_io *)ctsio);
 		break;
 	case CTL_ACTION_ERROR:
 	default:
 		mtx_unlock(&lun->lun_lock);
 		ctl_set_internal_failure(ctsio,
 					 /*sks_valid*/ 0,
 					 /*retry_count*/ 0);
 		ctl_done((union ctl_io *)ctsio);
 		break;
 	}
 	return (retval);
 }
 
 const struct ctl_cmd_entry *
 ctl_get_cmd_entry(struct ctl_scsiio *ctsio, int *sa)
 {
 	const struct ctl_cmd_entry *entry;
 	int service_action;
 
 	entry = &ctl_cmd_table[ctsio->cdb[0]];
 	if (sa)
 		*sa = ((entry->flags & CTL_CMD_FLAG_SA5) != 0);
 	if (entry->flags & CTL_CMD_FLAG_SA5) {
 		service_action = ctsio->cdb[1] & SERVICE_ACTION_MASK;
 		entry = &((const struct ctl_cmd_entry *)
 		    entry->execute)[service_action];
 	}
 	return (entry);
 }
 
 const struct ctl_cmd_entry *
 ctl_validate_command(struct ctl_scsiio *ctsio)
 {
 	const struct ctl_cmd_entry *entry;
 	int i, sa;
 	uint8_t diff;
 
 	entry = ctl_get_cmd_entry(ctsio, &sa);
 	if (entry->execute == NULL) {
 		if (sa)
 			ctl_set_invalid_field(ctsio,
 					      /*sks_valid*/ 1,
 					      /*command*/ 1,
 					      /*field*/ 1,
 					      /*bit_valid*/ 1,
 					      /*bit*/ 4);
 		else
 			ctl_set_invalid_opcode(ctsio);
 		ctl_done((union ctl_io *)ctsio);
 		return (NULL);
 	}
 	KASSERT(entry->length > 0,
 	    ("Not defined length for command 0x%02x/0x%02x",
 	     ctsio->cdb[0], ctsio->cdb[1]));
 	for (i = 1; i < entry->length; i++) {
 		diff = ctsio->cdb[i] & ~entry->usage[i - 1];
 		if (diff == 0)
 			continue;
 		ctl_set_invalid_field(ctsio,
 				      /*sks_valid*/ 1,
 				      /*command*/ 1,
 				      /*field*/ i,
 				      /*bit_valid*/ 1,
 				      /*bit*/ fls(diff) - 1);
 		ctl_done((union ctl_io *)ctsio);
 		return (NULL);
 	}
 	return (entry);
 }
 
 static int
 ctl_cmd_applicable(uint8_t lun_type, const struct ctl_cmd_entry *entry)
 {
 
 	switch (lun_type) {
 	case T_PROCESSOR:
 		if (((entry->flags & CTL_CMD_FLAG_OK_ON_PROC) == 0) &&
 		    ((entry->flags & CTL_CMD_FLAG_OK_ON_ALL_LUNS) == 0))
 			return (0);
 		break;
 	case T_DIRECT:
 		if (((entry->flags & CTL_CMD_FLAG_OK_ON_SLUN) == 0) &&
 		    ((entry->flags & CTL_CMD_FLAG_OK_ON_ALL_LUNS) == 0))
 			return (0);
 		break;
 	default:
 		return (0);
 	}
 	return (1);
 }
 
 static int
 ctl_scsiio(struct ctl_scsiio *ctsio)
 {
 	int retval;
 	const struct ctl_cmd_entry *entry;
 
 	retval = CTL_RETVAL_COMPLETE;
 
 	CTL_DEBUG_PRINT(("ctl_scsiio cdb[0]=%02X\n", ctsio->cdb[0]));
 
 	entry = ctl_get_cmd_entry(ctsio, NULL);
 
 	/*
 	 * If this I/O has been aborted, just send it straight to
 	 * ctl_done() without executing it.
 	 */
 	if (ctsio->io_hdr.flags & CTL_FLAG_ABORT) {
 		ctl_done((union ctl_io *)ctsio);
 		goto bailout;
 	}
 
 	/*
 	 * All the checks should have been handled by ctl_scsiio_precheck().
 	 * We should be clear now to just execute the I/O.
 	 */
 	retval = entry->execute(ctsio);
 
 bailout:
 	return (retval);
 }
 
 /*
  * Since we only implement one target right now, a bus reset simply resets
  * our single target.
  */
 static int
 ctl_bus_reset(struct ctl_softc *softc, union ctl_io *io)
 {
 	return(ctl_target_reset(softc, io, CTL_UA_BUS_RESET));
 }
 
 static int
 ctl_target_reset(struct ctl_softc *softc, union ctl_io *io,
 		 ctl_ua_type ua_type)
 {
 	struct ctl_lun *lun;
 	int retval;
 
 	if (!(io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC)) {
 		union ctl_ha_msg msg_info;
 
 		io->io_hdr.flags |= CTL_FLAG_SENT_2OTHER_SC;
 		msg_info.hdr.nexus = io->io_hdr.nexus;
 		if (ua_type==CTL_UA_TARG_RESET)
 			msg_info.task.task_action = CTL_TASK_TARGET_RESET;
 		else
 			msg_info.task.task_action = CTL_TASK_BUS_RESET;
 		msg_info.hdr.msg_type = CTL_MSG_MANAGE_TASKS;
 		msg_info.hdr.original_sc = NULL;
 		msg_info.hdr.serializing_sc = NULL;
 		if (CTL_HA_STATUS_SUCCESS != ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 		    (void *)&msg_info, sizeof(msg_info), 0)) {
 		}
 	}
 	retval = 0;
 
 	mtx_lock(&softc->ctl_lock);
 	STAILQ_FOREACH(lun, &softc->lun_list, links)
 		retval += ctl_lun_reset(lun, io, ua_type);
 	mtx_unlock(&softc->ctl_lock);
 
 	return (retval);
 }
 
 /*
  * The LUN should always be set.  The I/O is optional, and is used to
  * distinguish between I/Os sent by this initiator, and by other
  * initiators.  We set unit attention for initiators other than this one.
  * SAM-3 is vague on this point.  It does say that a unit attention should
  * be established for other initiators when a LUN is reset (see section
  * 5.7.3), but it doesn't specifically say that the unit attention should
  * be established for this particular initiator when a LUN is reset.  Here
  * is the relevant text, from SAM-3 rev 8:
  *
  * 5.7.2 When a SCSI initiator port aborts its own tasks
  *
  * When a SCSI initiator port causes its own task(s) to be aborted, no
  * notification that the task(s) have been aborted shall be returned to
  * the SCSI initiator port other than the completion response for the
  * command or task management function action that caused the task(s) to
  * be aborted and notification(s) associated with related effects of the
  * action (e.g., a reset unit attention condition).
  *
  * XXX KDM for now, we're setting unit attention for all initiators.
  */
 static int
 ctl_lun_reset(struct ctl_lun *lun, union ctl_io *io, ctl_ua_type ua_type)
 {
 	union ctl_io *xio;
 #if 0
 	uint32_t initidx;
 #endif
 #ifdef CTL_WITH_CA
 	int i;
 #endif
 
 	mtx_lock(&lun->lun_lock);
 	/*
 	 * Run through the OOA queue and abort each I/O.
 	 */
 #if 0
 	TAILQ_FOREACH((struct ctl_io_hdr *)xio, &lun->ooa_queue, ooa_links) {
 #endif
 	for (xio = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue); xio != NULL;
 	     xio = (union ctl_io *)TAILQ_NEXT(&xio->io_hdr, ooa_links)) {
 		xio->io_hdr.flags |= CTL_FLAG_ABORT | CTL_FLAG_ABORT_STATUS;
 	}
 
 	/*
 	 * This version sets unit attention for every
 	 */
 #if 0
 	initidx = ctl_get_initindex(&io->io_hdr.nexus);
 	ctl_est_ua_all(lun, initidx, ua_type);
 #else
 	ctl_est_ua_all(lun, -1, ua_type);
 #endif
 
 	/*
 	 * A reset (any kind, really) clears reservations established with
 	 * RESERVE/RELEASE.  It does not clear reservations established
 	 * with PERSISTENT RESERVE OUT, but we don't support that at the
 	 * moment anyway.  See SPC-2, section 5.6.  SPC-3 doesn't address
 	 * reservations made with the RESERVE/RELEASE commands, because
 	 * those commands are obsolete in SPC-3.
 	 */
 	lun->flags &= ~CTL_LUN_RESERVED;
 
 #ifdef CTL_WITH_CA
 	for (i = 0; i < CTL_MAX_INITIATORS; i++)
 		ctl_clear_mask(lun->have_ca, i);
 #endif
 	mtx_unlock(&lun->lun_lock);
 
 	return (0);
 }
 
 static void
 ctl_abort_tasks_lun(struct ctl_lun *lun, uint32_t targ_port, uint32_t init_id,
     int other_sc)
 {
 	union ctl_io *xio;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 
 	/*
 	 * Run through the OOA queue and attempt to find the given I/O.
 	 * The target port, initiator ID, tag type and tag number have to
 	 * match the values that we got from the initiator.  If we have an
 	 * untagged command to abort, simply abort the first untagged command
 	 * we come to.  We only allow one untagged command at a time of course.
 	 */
 	for (xio = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue); xio != NULL;
 	     xio = (union ctl_io *)TAILQ_NEXT(&xio->io_hdr, ooa_links)) {
 
 		if ((targ_port == UINT32_MAX ||
 		     targ_port == xio->io_hdr.nexus.targ_port) &&
 		    (init_id == UINT32_MAX ||
 		     init_id == xio->io_hdr.nexus.initid.id)) {
 			if (targ_port != xio->io_hdr.nexus.targ_port ||
 			    init_id != xio->io_hdr.nexus.initid.id)
 				xio->io_hdr.flags |= CTL_FLAG_ABORT_STATUS;
 			xio->io_hdr.flags |= CTL_FLAG_ABORT;
 			if (!other_sc && !(lun->flags & CTL_LUN_PRIMARY_SC)) {
 				union ctl_ha_msg msg_info;
 
 				msg_info.hdr.nexus = xio->io_hdr.nexus;
 				msg_info.task.task_action = CTL_TASK_ABORT_TASK;
 				msg_info.task.tag_num = xio->scsiio.tag_num;
 				msg_info.task.tag_type = xio->scsiio.tag_type;
 				msg_info.hdr.msg_type = CTL_MSG_MANAGE_TASKS;
 				msg_info.hdr.original_sc = NULL;
 				msg_info.hdr.serializing_sc = NULL;
 				ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 				    (void *)&msg_info, sizeof(msg_info), 0);
 			}
 		}
 	}
 }
 
 static int
 ctl_abort_task_set(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_lun *lun;
 	uint32_t targ_lun;
 
 	/*
 	 * Look up the LUN.
 	 */
 	targ_lun = io->io_hdr.nexus.targ_mapped_lun;
 	mtx_lock(&softc->ctl_lock);
 	if ((targ_lun < CTL_MAX_LUNS) && (softc->ctl_luns[targ_lun] != NULL))
 		lun = softc->ctl_luns[targ_lun];
 	else {
 		mtx_unlock(&softc->ctl_lock);
 		return (1);
 	}
 
 	mtx_lock(&lun->lun_lock);
 	mtx_unlock(&softc->ctl_lock);
 	if (io->taskio.task_action == CTL_TASK_ABORT_TASK_SET) {
 		ctl_abort_tasks_lun(lun, io->io_hdr.nexus.targ_port,
 		    io->io_hdr.nexus.initid.id,
 		    (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC) != 0);
 	} else { /* CTL_TASK_CLEAR_TASK_SET */
 		ctl_abort_tasks_lun(lun, UINT32_MAX, UINT32_MAX,
 		    (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC) != 0);
 	}
 	mtx_unlock(&lun->lun_lock);
 	return (0);
 }
 
 static int
 ctl_i_t_nexus_reset(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_lun *lun;
 	uint32_t initidx, residx;
 
 	initidx = ctl_get_initindex(&io->io_hdr.nexus);
 	residx = ctl_get_resindex(&io->io_hdr.nexus);
 	mtx_lock(&softc->ctl_lock);
 	STAILQ_FOREACH(lun, &softc->lun_list, links) {
 		mtx_lock(&lun->lun_lock);
 		ctl_abort_tasks_lun(lun, io->io_hdr.nexus.targ_port,
 		    io->io_hdr.nexus.initid.id,
 		    (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC) != 0);
 #ifdef CTL_WITH_CA
 		ctl_clear_mask(lun->have_ca, initidx);
 #endif
 		if ((lun->flags & CTL_LUN_RESERVED) && (lun->res_idx == residx))
 			lun->flags &= ~CTL_LUN_RESERVED;
 		ctl_est_ua(lun, initidx, CTL_UA_I_T_NEXUS_LOSS);
 		mtx_unlock(&lun->lun_lock);
 	}
 	mtx_unlock(&softc->ctl_lock);
 	return (0);
 }
 
 static int
 ctl_abort_task(union ctl_io *io)
 {
 	union ctl_io *xio;
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 #if 0
 	struct sbuf sb;
 	char printbuf[128];
 #endif
 	int found;
 	uint32_t targ_lun;
 
 	softc = control_softc;
 	found = 0;
 
 	/*
 	 * Look up the LUN.
 	 */
 	targ_lun = io->io_hdr.nexus.targ_mapped_lun;
 	mtx_lock(&softc->ctl_lock);
 	if ((targ_lun < CTL_MAX_LUNS)
 	 && (softc->ctl_luns[targ_lun] != NULL))
 		lun = softc->ctl_luns[targ_lun];
 	else {
 		mtx_unlock(&softc->ctl_lock);
 		return (1);
 	}
 
 #if 0
 	printf("ctl_abort_task: called for lun %lld, tag %d type %d\n",
 	       lun->lun, io->taskio.tag_num, io->taskio.tag_type);
 #endif
 
 	mtx_lock(&lun->lun_lock);
 	mtx_unlock(&softc->ctl_lock);
 	/*
 	 * Run through the OOA queue and attempt to find the given I/O.
 	 * The target port, initiator ID, tag type and tag number have to
 	 * match the values that we got from the initiator.  If we have an
 	 * untagged command to abort, simply abort the first untagged command
 	 * we come to.  We only allow one untagged command at a time of course.
 	 */
 #if 0
 	TAILQ_FOREACH((struct ctl_io_hdr *)xio, &lun->ooa_queue, ooa_links) {
 #endif
 	for (xio = (union ctl_io *)TAILQ_FIRST(&lun->ooa_queue); xio != NULL;
 	     xio = (union ctl_io *)TAILQ_NEXT(&xio->io_hdr, ooa_links)) {
 #if 0
 		sbuf_new(&sb, printbuf, sizeof(printbuf), SBUF_FIXEDLEN);
 
 		sbuf_printf(&sb, "LUN %lld tag %d type %d%s%s%s%s: ",
 			    lun->lun, xio->scsiio.tag_num,
 			    xio->scsiio.tag_type,
 			    (xio->io_hdr.blocked_links.tqe_prev
 			    == NULL) ? "" : " BLOCKED",
 			    (xio->io_hdr.flags &
 			    CTL_FLAG_DMA_INPROG) ? " DMA" : "",
 			    (xio->io_hdr.flags &
 			    CTL_FLAG_ABORT) ? " ABORT" : "",
 			    (xio->io_hdr.flags &
 			    CTL_FLAG_IS_WAS_ON_RTR ? " RTR" : ""));
 		ctl_scsi_command_string(&xio->scsiio, NULL, &sb);
 		sbuf_finish(&sb);
 		printf("%s\n", sbuf_data(&sb));
 #endif
 
 		if ((xio->io_hdr.nexus.targ_port == io->io_hdr.nexus.targ_port)
 		 && (xio->io_hdr.nexus.initid.id ==
 		     io->io_hdr.nexus.initid.id)) {
 			/*
 			 * If the abort says that the task is untagged, the
 			 * task in the queue must be untagged.  Otherwise,
 			 * we just check to see whether the tag numbers
 			 * match.  This is because the QLogic firmware
 			 * doesn't pass back the tag type in an abort
 			 * request.
 			 */
 #if 0
 			if (((xio->scsiio.tag_type == CTL_TAG_UNTAGGED)
 			  && (io->taskio.tag_type == CTL_TAG_UNTAGGED))
 			 || (xio->scsiio.tag_num == io->taskio.tag_num)) {
 #endif
 			/*
 			 * XXX KDM we've got problems with FC, because it
 			 * doesn't send down a tag type with aborts.  So we
 			 * can only really go by the tag number...
 			 * This may cause problems with parallel SCSI.
 			 * Need to figure that out!!
 			 */
 			if (xio->scsiio.tag_num == io->taskio.tag_num) {
 				xio->io_hdr.flags |= CTL_FLAG_ABORT;
 				found = 1;
 				if ((io->io_hdr.flags &
 				     CTL_FLAG_FROM_OTHER_SC) == 0 &&
 				    !(lun->flags & CTL_LUN_PRIMARY_SC)) {
 					union ctl_ha_msg msg_info;
 
 					io->io_hdr.flags |=
 					                CTL_FLAG_SENT_2OTHER_SC;
 					msg_info.hdr.nexus = io->io_hdr.nexus;
 					msg_info.task.task_action =
 						CTL_TASK_ABORT_TASK;
 					msg_info.task.tag_num =
 						io->taskio.tag_num;
 					msg_info.task.tag_type =
 						io->taskio.tag_type;
 					msg_info.hdr.msg_type =
 						CTL_MSG_MANAGE_TASKS;
 					msg_info.hdr.original_sc = NULL;
 					msg_info.hdr.serializing_sc = NULL;
 #if 0
 					printf("Sent Abort to other side\n");
 #endif
 					if (CTL_HA_STATUS_SUCCESS !=
 					        ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 		    				(void *)&msg_info,
 						sizeof(msg_info), 0)) {
 					}
 				}
 #if 0
 				printf("ctl_abort_task: found I/O to abort\n");
 #endif
 				break;
 			}
 		}
 	}
 	mtx_unlock(&lun->lun_lock);
 
 	if (found == 0) {
 		/*
 		 * This isn't really an error.  It's entirely possible for
 		 * the abort and command completion to cross on the wire.
 		 * This is more of an informative/diagnostic error.
 		 */
 #if 0
 		printf("ctl_abort_task: ABORT sent for nonexistent I/O: "
 		       "%d:%d:%d:%d tag %d type %d\n",
 		       io->io_hdr.nexus.initid.id,
 		       io->io_hdr.nexus.targ_port,
 		       io->io_hdr.nexus.targ_target.id,
 		       io->io_hdr.nexus.targ_lun, io->taskio.tag_num,
 		       io->taskio.tag_type);
 #endif
 	}
 	return (0);
 }
 
 static void
 ctl_run_task(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	int retval = 1;
 	const char *task_desc;
 
 	CTL_DEBUG_PRINT(("ctl_run_task\n"));
 
 	KASSERT(io->io_hdr.io_type == CTL_IO_TASK,
 	    ("ctl_run_task: Unextected io_type %d\n",
 	     io->io_hdr.io_type));
 
 	task_desc = ctl_scsi_task_string(&io->taskio);
 	if (task_desc != NULL) {
 #ifdef NEEDTOPORT
 		csevent_log(CSC_CTL | CSC_SHELF_SW |
 			    CTL_TASK_REPORT,
 			    csevent_LogType_Trace,
 			    csevent_Severity_Information,
 			    csevent_AlertLevel_Green,
 			    csevent_FRU_Firmware,
 			    csevent_FRU_Unknown,
 			    "CTL: received task: %s",task_desc);
 #endif
 	} else {
 #ifdef NEEDTOPORT
 		csevent_log(CSC_CTL | CSC_SHELF_SW |
 			    CTL_TASK_REPORT,
 			    csevent_LogType_Trace,
 			    csevent_Severity_Information,
 			    csevent_AlertLevel_Green,
 			    csevent_FRU_Firmware,
 			    csevent_FRU_Unknown,
 			    "CTL: received unknown task "
 			    "type: %d (%#x)",
 			    io->taskio.task_action,
 			    io->taskio.task_action);
 #endif
 	}
 	switch (io->taskio.task_action) {
 	case CTL_TASK_ABORT_TASK:
 		retval = ctl_abort_task(io);
 		break;
 	case CTL_TASK_ABORT_TASK_SET:
 	case CTL_TASK_CLEAR_TASK_SET:
 		retval = ctl_abort_task_set(io);
 		break;
 	case CTL_TASK_CLEAR_ACA:
 		break;
 	case CTL_TASK_I_T_NEXUS_RESET:
 		retval = ctl_i_t_nexus_reset(io);
 		break;
 	case CTL_TASK_LUN_RESET: {
 		struct ctl_lun *lun;
 		uint32_t targ_lun;
 
 		targ_lun = io->io_hdr.nexus.targ_mapped_lun;
 		mtx_lock(&softc->ctl_lock);
 		if ((targ_lun < CTL_MAX_LUNS)
 		 && (softc->ctl_luns[targ_lun] != NULL))
 			lun = softc->ctl_luns[targ_lun];
 		else {
 			mtx_unlock(&softc->ctl_lock);
 			retval = 1;
 			break;
 		}
 
 		if (!(io->io_hdr.flags &
 		    CTL_FLAG_FROM_OTHER_SC)) {
 			union ctl_ha_msg msg_info;
 
 			io->io_hdr.flags |=
 				CTL_FLAG_SENT_2OTHER_SC;
 			msg_info.hdr.msg_type =
 				CTL_MSG_MANAGE_TASKS;
 			msg_info.hdr.nexus = io->io_hdr.nexus;
 			msg_info.task.task_action =
 				CTL_TASK_LUN_RESET;
 			msg_info.hdr.original_sc = NULL;
 			msg_info.hdr.serializing_sc = NULL;
 			if (CTL_HA_STATUS_SUCCESS !=
 			    ctl_ha_msg_send(CTL_HA_CHAN_CTL,
 			    (void *)&msg_info,
 			    sizeof(msg_info), 0)) {
 			}
 		}
 
 		retval = ctl_lun_reset(lun, io,
 				       CTL_UA_LUN_RESET);
 		mtx_unlock(&softc->ctl_lock);
 		break;
 	}
 	case CTL_TASK_TARGET_RESET:
 		retval = ctl_target_reset(softc, io, CTL_UA_TARG_RESET);
 		break;
 	case CTL_TASK_BUS_RESET:
 		retval = ctl_bus_reset(softc, io);
 		break;
 	case CTL_TASK_PORT_LOGIN:
 		break;
 	case CTL_TASK_PORT_LOGOUT:
 		break;
 	default:
 		printf("ctl_run_task: got unknown task management event %d\n",
 		       io->taskio.task_action);
 		break;
 	}
 	if (retval == 0)
 		io->io_hdr.status = CTL_SUCCESS;
 	else
 		io->io_hdr.status = CTL_ERROR;
 	ctl_done(io);
 }
 
 /*
  * For HA operation.  Handle commands that come in from the other
  * controller.
  */
 static void
 ctl_handle_isc(union ctl_io *io)
 {
 	int free_io;
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	uint32_t targ_lun;
 
 	softc = control_softc;
 
 	targ_lun = io->io_hdr.nexus.targ_mapped_lun;
 	lun = softc->ctl_luns[targ_lun];
 
 	switch (io->io_hdr.msg_type) {
 	case CTL_MSG_SERIALIZE:
 		free_io = ctl_serialize_other_sc_cmd(&io->scsiio);
 		break;
 	case CTL_MSG_R2R: {
 		const struct ctl_cmd_entry *entry;
 
 		/*
 		 * This is only used in SER_ONLY mode.
 		 */
 		free_io = 0;
 		entry = ctl_get_cmd_entry(&io->scsiio, NULL);
 		mtx_lock(&lun->lun_lock);
 		if (ctl_scsiio_lun_check(lun,
 		    entry, (struct ctl_scsiio *)io) != 0) {
 			mtx_unlock(&lun->lun_lock);
 			ctl_done(io);
 			break;
 		}
 		io->io_hdr.flags |= CTL_FLAG_IS_WAS_ON_RTR;
 		mtx_unlock(&lun->lun_lock);
 		ctl_enqueue_rtr(io);
 		break;
 	}
 	case CTL_MSG_FINISH_IO:
 		if (softc->ha_mode == CTL_HA_MODE_XFER) {
 			free_io = 0;
 			ctl_done(io);
 		} else {
 			free_io = 1;
 			mtx_lock(&lun->lun_lock);
 			TAILQ_REMOVE(&lun->ooa_queue, &io->io_hdr,
 				     ooa_links);
 			ctl_check_blocked(lun);
 			mtx_unlock(&lun->lun_lock);
 		}
 		break;
 	case CTL_MSG_PERS_ACTION:
 		ctl_hndl_per_res_out_on_other_sc(
 			(union ctl_ha_msg *)&io->presio.pr_msg);
 		free_io = 1;
 		break;
 	case CTL_MSG_BAD_JUJU:
 		free_io = 0;
 		ctl_done(io);
 		break;
 	case CTL_MSG_DATAMOVE:
 		/* Only used in XFER mode */
 		free_io = 0;
 		ctl_datamove_remote(io);
 		break;
 	case CTL_MSG_DATAMOVE_DONE:
 		/* Only used in XFER mode */
 		free_io = 0;
 		io->scsiio.be_move_done(io);
 		break;
 	default:
 		free_io = 1;
 		printf("%s: Invalid message type %d\n",
 		       __func__, io->io_hdr.msg_type);
 		break;
 	}
 	if (free_io)
 		ctl_free_io(io);
 
 }
 
 
 /*
  * Returns the match type in the case of a match, or CTL_LUN_PAT_NONE if
  * there is no match.
  */
 static ctl_lun_error_pattern
 ctl_cmd_pattern_match(struct ctl_scsiio *ctsio, struct ctl_error_desc *desc)
 {
 	const struct ctl_cmd_entry *entry;
 	ctl_lun_error_pattern filtered_pattern, pattern;
 
 	pattern = desc->error_pattern;
 
 	/*
 	 * XXX KDM we need more data passed into this function to match a
 	 * custom pattern, and we actually need to implement custom pattern
 	 * matching.
 	 */
 	if (pattern & CTL_LUN_PAT_CMD)
 		return (CTL_LUN_PAT_CMD);
 
 	if ((pattern & CTL_LUN_PAT_MASK) == CTL_LUN_PAT_ANY)
 		return (CTL_LUN_PAT_ANY);
 
 	entry = ctl_get_cmd_entry(ctsio, NULL);
 
 	filtered_pattern = entry->pattern & pattern;
 
 	/*
 	 * If the user requested specific flags in the pattern (e.g.
 	 * CTL_LUN_PAT_RANGE), make sure the command supports all of those
 	 * flags.
 	 *
 	 * If the user did not specify any flags, it doesn't matter whether
 	 * or not the command supports the flags.
 	 */
 	if ((filtered_pattern & ~CTL_LUN_PAT_MASK) !=
 	     (pattern & ~CTL_LUN_PAT_MASK))
 		return (CTL_LUN_PAT_NONE);
 
 	/*
 	 * If the user asked for a range check, see if the requested LBA
 	 * range overlaps with this command's LBA range.
 	 */
 	if (filtered_pattern & CTL_LUN_PAT_RANGE) {
 		uint64_t lba1;
 		uint64_t len1;
 		ctl_action action;
 		int retval;
 
 		retval = ctl_get_lba_len((union ctl_io *)ctsio, &lba1, &len1);
 		if (retval != 0)
 			return (CTL_LUN_PAT_NONE);
 
 		action = ctl_extent_check_lba(lba1, len1, desc->lba_range.lba,
 					      desc->lba_range.len, FALSE);
 		/*
 		 * A "pass" means that the LBA ranges don't overlap, so
 		 * this doesn't match the user's range criteria.
 		 */
 		if (action == CTL_ACTION_PASS)
 			return (CTL_LUN_PAT_NONE);
 	}
 
 	return (filtered_pattern);
 }
 
 static void
 ctl_inject_error(struct ctl_lun *lun, union ctl_io *io)
 {
 	struct ctl_error_desc *desc, *desc2;
 
 	mtx_assert(&lun->lun_lock, MA_OWNED);
 
 	STAILQ_FOREACH_SAFE(desc, &lun->error_list, links, desc2) {
 		ctl_lun_error_pattern pattern;
 		/*
 		 * Check to see whether this particular command matches
 		 * the pattern in the descriptor.
 		 */
 		pattern = ctl_cmd_pattern_match(&io->scsiio, desc);
 		if ((pattern & CTL_LUN_PAT_MASK) == CTL_LUN_PAT_NONE)
 			continue;
 
 		switch (desc->lun_error & CTL_LUN_INJ_TYPE) {
 		case CTL_LUN_INJ_ABORTED:
 			ctl_set_aborted(&io->scsiio);
 			break;
 		case CTL_LUN_INJ_MEDIUM_ERR:
 			ctl_set_medium_error(&io->scsiio);
 			break;
 		case CTL_LUN_INJ_UA:
 			/* 29h/00h  POWER ON, RESET, OR BUS DEVICE RESET
 			 * OCCURRED */
 			ctl_set_ua(&io->scsiio, 0x29, 0x00);
 			break;
 		case CTL_LUN_INJ_CUSTOM:
 			/*
 			 * We're assuming the user knows what he is doing.
 			 * Just copy the sense information without doing
 			 * checks.
 			 */
 			bcopy(&desc->custom_sense, &io->scsiio.sense_data,
 			      MIN(sizeof(desc->custom_sense),
 				  sizeof(io->scsiio.sense_data)));
 			io->scsiio.scsi_status = SCSI_STATUS_CHECK_COND;
 			io->scsiio.sense_len = SSD_FULL_SIZE;
 			io->io_hdr.status = CTL_SCSI_ERROR | CTL_AUTOSENSE;
 			break;
 		case CTL_LUN_INJ_NONE:
 		default:
 			/*
 			 * If this is an error injection type we don't know
 			 * about, clear the continuous flag (if it is set)
 			 * so it will get deleted below.
 			 */
 			desc->lun_error &= ~CTL_LUN_INJ_CONTINUOUS;
 			break;
 		}
 		/*
 		 * By default, each error injection action is a one-shot
 		 */
 		if (desc->lun_error & CTL_LUN_INJ_CONTINUOUS)
 			continue;
 
 		STAILQ_REMOVE(&lun->error_list, desc, ctl_error_desc, links);
 
 		free(desc, M_CTL);
 	}
 }
 
 #ifdef CTL_IO_DELAY
 static void
 ctl_datamove_timer_wakeup(void *arg)
 {
 	union ctl_io *io;
 
 	io = (union ctl_io *)arg;
 
 	ctl_datamove(io);
 }
 #endif /* CTL_IO_DELAY */
 
 void
 ctl_datamove(union ctl_io *io)
 {
 	void (*fe_datamove)(union ctl_io *io);
 
 	mtx_assert(&control_softc->ctl_lock, MA_NOTOWNED);
 
 	CTL_DEBUG_PRINT(("ctl_datamove\n"));
 
 #ifdef CTL_TIME_IO
 	if ((time_uptime - io->io_hdr.start_time) > ctl_time_io_secs) {
 		char str[256];
 		char path_str[64];
 		struct sbuf sb;
 
 		ctl_scsi_path_string(io, path_str, sizeof(path_str));
 		sbuf_new(&sb, str, sizeof(str), SBUF_FIXEDLEN);
 
 		sbuf_cat(&sb, path_str);
 		switch (io->io_hdr.io_type) {
 		case CTL_IO_SCSI:
 			ctl_scsi_command_string(&io->scsiio, NULL, &sb);
 			sbuf_printf(&sb, "\n");
 			sbuf_cat(&sb, path_str);
 			sbuf_printf(&sb, "Tag: 0x%04x, type %d\n",
 				    io->scsiio.tag_num, io->scsiio.tag_type);
 			break;
 		case CTL_IO_TASK:
 			sbuf_printf(&sb, "Task I/O type: %d, Tag: 0x%04x, "
 				    "Tag Type: %d\n", io->taskio.task_action,
 				    io->taskio.tag_num, io->taskio.tag_type);
 			break;
 		default:
 			printf("Invalid CTL I/O type %d\n", io->io_hdr.io_type);
 			panic("Invalid CTL I/O type %d\n", io->io_hdr.io_type);
 			break;
 		}
 		sbuf_cat(&sb, path_str);
 		sbuf_printf(&sb, "ctl_datamove: %jd seconds\n",
 			    (intmax_t)time_uptime - io->io_hdr.start_time);
 		sbuf_finish(&sb);
 		printf("%s", sbuf_data(&sb));
 	}
 #endif /* CTL_TIME_IO */
 
 #ifdef CTL_IO_DELAY
 	if (io->io_hdr.flags & CTL_FLAG_DELAY_DONE) {
 		struct ctl_lun *lun;
 
 		lun =(struct ctl_lun *)io->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 		io->io_hdr.flags &= ~CTL_FLAG_DELAY_DONE;
 	} else {
 		struct ctl_lun *lun;
 
 		lun =(struct ctl_lun *)io->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 		if ((lun != NULL)
 		 && (lun->delay_info.datamove_delay > 0)) {
 			struct callout *callout;
 
 			callout = (struct callout *)&io->io_hdr.timer_bytes;
 			callout_init(callout, /*mpsafe*/ 1);
 			io->io_hdr.flags |= CTL_FLAG_DELAY_DONE;
 			callout_reset(callout,
 				      lun->delay_info.datamove_delay * hz,
 				      ctl_datamove_timer_wakeup, io);
 			if (lun->delay_info.datamove_type ==
 			    CTL_DELAY_TYPE_ONESHOT)
 				lun->delay_info.datamove_delay = 0;
 			return;
 		}
 	}
 #endif
 
 	/*
 	 * This command has been aborted.  Set the port status, so we fail
 	 * the data move.
 	 */
 	if (io->io_hdr.flags & CTL_FLAG_ABORT) {
 		printf("ctl_datamove: tag 0x%04x on (%ju:%d:%ju:%d) aborted\n",
 		       io->scsiio.tag_num,(uintmax_t)io->io_hdr.nexus.initid.id,
 		       io->io_hdr.nexus.targ_port,
 		       (uintmax_t)io->io_hdr.nexus.targ_target.id,
 		       io->io_hdr.nexus.targ_lun);
 		io->io_hdr.port_status = 31337;
 		/*
 		 * Note that the backend, in this case, will get the
 		 * callback in its context.  In other cases it may get
 		 * called in the frontend's interrupt thread context.
 		 */
 		io->scsiio.be_move_done(io);
 		return;
 	}
 
 	/* Don't confuse frontend with zero length data move. */
 	if (io->scsiio.kern_data_len == 0) {
 		io->scsiio.be_move_done(io);
 		return;
 	}
 
 	/*
 	 * If we're in XFER mode and this I/O is from the other shelf
 	 * controller, we need to send the DMA to the other side to
 	 * actually transfer the data to/from the host.  In serialize only
 	 * mode the transfer happens below CTL and ctl_datamove() is only
 	 * called on the machine that originally received the I/O.
 	 */
 	if ((control_softc->ha_mode == CTL_HA_MODE_XFER)
 	 && (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC)) {
 		union ctl_ha_msg msg;
 		uint32_t sg_entries_sent;
 		int do_sg_copy;
 		int i;
 
 		memset(&msg, 0, sizeof(msg));
 		msg.hdr.msg_type = CTL_MSG_DATAMOVE;
 		msg.hdr.original_sc = io->io_hdr.original_sc;
 		msg.hdr.serializing_sc = io;
 		msg.hdr.nexus = io->io_hdr.nexus;
 		msg.dt.flags = io->io_hdr.flags;
 		/*
 		 * We convert everything into a S/G list here.  We can't
 		 * pass by reference, only by value between controllers.
 		 * So we can't pass a pointer to the S/G list, only as many
 		 * S/G entries as we can fit in here.  If it's possible for
 		 * us to get more than CTL_HA_MAX_SG_ENTRIES S/G entries,
 		 * then we need to break this up into multiple transfers.
 		 */
 		if (io->scsiio.kern_sg_entries == 0) {
 			msg.dt.kern_sg_entries = 1;
 			/*
 			 * If this is in cached memory, flush the cache
 			 * before we send the DMA request to the other
 			 * controller.  We want to do this in either the
 			 * read or the write case.  The read case is
 			 * straightforward.  In the write case, we want to
 			 * make sure nothing is in the local cache that
 			 * could overwrite the DMAed data.
 			 */
 			if ((io->io_hdr.flags & CTL_FLAG_NO_DATASYNC) == 0) {
 				/*
 				 * XXX KDM use bus_dmamap_sync() here.
 				 */
 			}
 
 			/*
 			 * Convert to a physical address if this is a
 			 * virtual address.
 			 */
 			if (io->io_hdr.flags & CTL_FLAG_BUS_ADDR) {
 				msg.dt.sg_list[0].addr =
 					io->scsiio.kern_data_ptr;
 			} else {
 				/*
 				 * XXX KDM use busdma here!
 				 */
 #if 0
 				msg.dt.sg_list[0].addr = (void *)
 					vtophys(io->scsiio.kern_data_ptr);
 #endif
 			}
 
 			msg.dt.sg_list[0].len = io->scsiio.kern_data_len;
 			do_sg_copy = 0;
 		} else {
 			struct ctl_sg_entry *sgl;
 
 			do_sg_copy = 1;
 			msg.dt.kern_sg_entries = io->scsiio.kern_sg_entries;
 			sgl = (struct ctl_sg_entry *)io->scsiio.kern_data_ptr;
 			if ((io->io_hdr.flags & CTL_FLAG_NO_DATASYNC) == 0) {
 				/*
 				 * XXX KDM use bus_dmamap_sync() here.
 				 */
 			}
 		}
 
 		msg.dt.kern_data_len = io->scsiio.kern_data_len;
 		msg.dt.kern_total_len = io->scsiio.kern_total_len;
 		msg.dt.kern_data_resid = io->scsiio.kern_data_resid;
 		msg.dt.kern_rel_offset = io->scsiio.kern_rel_offset;
 		msg.dt.sg_sequence = 0;
 
 		/*
 		 * Loop until we've sent all of the S/G entries.  On the
 		 * other end, we'll recompose these S/G entries into one
 		 * contiguous list before passing it to the
 		 */
 		for (sg_entries_sent = 0; sg_entries_sent <
 		     msg.dt.kern_sg_entries; msg.dt.sg_sequence++) {
 			msg.dt.cur_sg_entries = MIN((sizeof(msg.dt.sg_list)/
 				sizeof(msg.dt.sg_list[0])),
 				msg.dt.kern_sg_entries - sg_entries_sent);
 
 			if (do_sg_copy != 0) {
 				struct ctl_sg_entry *sgl;
 				int j;
 
 				sgl = (struct ctl_sg_entry *)
 					io->scsiio.kern_data_ptr;
 				/*
 				 * If this is in cached memory, flush the cache
 				 * before we send the DMA request to the other
 				 * controller.  We want to do this in either
 				 * the * read or the write case.  The read
 				 * case is straightforward.  In the write
 				 * case, we want to make sure nothing is
 				 * in the local cache that could overwrite
 				 * the DMAed data.
 				 */
 
 				for (i = sg_entries_sent, j = 0;
 				     i < msg.dt.cur_sg_entries; i++, j++) {
 					if ((io->io_hdr.flags &
 					     CTL_FLAG_NO_DATASYNC) == 0) {
 						/*
 						 * XXX KDM use bus_dmamap_sync()
 						 */
 					}
 					if ((io->io_hdr.flags &
 					     CTL_FLAG_BUS_ADDR) == 0) {
 						/*
 						 * XXX KDM use busdma.
 						 */
 #if 0
 						msg.dt.sg_list[j].addr =(void *)
 						       vtophys(sgl[i].addr);
 #endif
 					} else {
 						msg.dt.sg_list[j].addr =
 							sgl[i].addr;
 					}
 					msg.dt.sg_list[j].len = sgl[i].len;
 				}
 			}
 
 			sg_entries_sent += msg.dt.cur_sg_entries;
 			if (sg_entries_sent >= msg.dt.kern_sg_entries)
 				msg.dt.sg_last = 1;
 			else
 				msg.dt.sg_last = 0;
 
 			/*
 			 * XXX KDM drop and reacquire the lock here?
 			 */
 			if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg,
 			    sizeof(msg), 0) > CTL_HA_STATUS_SUCCESS) {
 				/*
 				 * XXX do something here.
 				 */
 			}
 
 			msg.dt.sent_sg_entries = sg_entries_sent;
 		}
 		io->io_hdr.flags &= ~CTL_FLAG_IO_ACTIVE;
 		if (io->io_hdr.flags & CTL_FLAG_FAILOVER)
 			ctl_failover_io(io, /*have_lock*/ 0);
 
 	} else {
 
 		/*
 		 * Lookup the fe_datamove() function for this particular
 		 * front end.
 		 */
 		fe_datamove =
 		    control_softc->ctl_ports[ctl_port_idx(io->io_hdr.nexus.targ_port)]->fe_datamove;
 
 		fe_datamove(io);
 	}
 }
 
 static void
 ctl_send_datamove_done(union ctl_io *io, int have_lock)
 {
 	union ctl_ha_msg msg;
 	int isc_status;
 
 	memset(&msg, 0, sizeof(msg));
 
 	msg.hdr.msg_type = CTL_MSG_DATAMOVE_DONE;
 	msg.hdr.original_sc = io;
 	msg.hdr.serializing_sc = io->io_hdr.serializing_sc;
 	msg.hdr.nexus = io->io_hdr.nexus;
 	msg.hdr.status = io->io_hdr.status;
 	msg.scsi.tag_num = io->scsiio.tag_num;
 	msg.scsi.tag_type = io->scsiio.tag_type;
 	msg.scsi.scsi_status = io->scsiio.scsi_status;
 	memcpy(&msg.scsi.sense_data, &io->scsiio.sense_data,
 	       sizeof(io->scsiio.sense_data));
 	msg.scsi.sense_len = io->scsiio.sense_len;
 	msg.scsi.sense_residual = io->scsiio.sense_residual;
 	msg.scsi.fetd_status = io->io_hdr.port_status;
 	msg.scsi.residual = io->scsiio.residual;
 	io->io_hdr.flags &= ~CTL_FLAG_IO_ACTIVE;
 
 	if (io->io_hdr.flags & CTL_FLAG_FAILOVER) {
 		ctl_failover_io(io, /*have_lock*/ have_lock);
 		return;
 	}
 
 	isc_status = ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg, sizeof(msg), 0);
 	if (isc_status > CTL_HA_STATUS_SUCCESS) {
 		/* XXX do something if this fails */
 	}
 
 }
 
 /*
  * The DMA to the remote side is done, now we need to tell the other side
  * we're done so it can continue with its data movement.
  */
 static void
 ctl_datamove_remote_write_cb(struct ctl_ha_dt_req *rq)
 {
 	union ctl_io *io;
 
 	io = rq->context;
 
 	if (rq->ret != CTL_HA_STATUS_SUCCESS) {
 		printf("%s: ISC DMA write failed with error %d", __func__,
 		       rq->ret);
 		ctl_set_internal_failure(&io->scsiio,
 					 /*sks_valid*/ 1,
 					 /*retry_count*/ rq->ret);
 	}
 
 	ctl_dt_req_free(rq);
 
 	/*
 	 * In this case, we had to malloc the memory locally.  Free it.
 	 */
 	if ((io->io_hdr.flags & CTL_FLAG_AUTO_MIRROR) == 0) {
 		int i;
 		for (i = 0; i < io->scsiio.kern_sg_entries; i++)
 			free(io->io_hdr.local_sglist[i].addr, M_CTL);
 	}
 	/*
 	 * The data is in local and remote memory, so now we need to send
 	 * status (good or back) back to the other side.
 	 */
 	ctl_send_datamove_done(io, /*have_lock*/ 0);
 }
 
 /*
  * We've moved the data from the host/controller into local memory.  Now we
  * need to push it over to the remote controller's memory.
  */
 static int
 ctl_datamove_remote_dm_write_cb(union ctl_io *io)
 {
 	int retval;
 
 	retval = 0;
 
 	retval = ctl_datamove_remote_xfer(io, CTL_HA_DT_CMD_WRITE,
 					  ctl_datamove_remote_write_cb);
 
 	return (retval);
 }
 
 static void
 ctl_datamove_remote_write(union ctl_io *io)
 {
 	int retval;
 	void (*fe_datamove)(union ctl_io *io);
 
 	/*
 	 * - Get the data from the host/HBA into local memory.
 	 * - DMA memory from the local controller to the remote controller.
 	 * - Send status back to the remote controller.
 	 */
 
 	retval = ctl_datamove_remote_sgl_setup(io);
 	if (retval != 0)
 		return;
 
 	/* Switch the pointer over so the FETD knows what to do */
 	io->scsiio.kern_data_ptr = (uint8_t *)io->io_hdr.local_sglist;
 
 	/*
 	 * Use a custom move done callback, since we need to send completion
 	 * back to the other controller, not to the backend on this side.
 	 */
 	io->scsiio.be_move_done = ctl_datamove_remote_dm_write_cb;
 
 	fe_datamove = control_softc->ctl_ports[ctl_port_idx(io->io_hdr.nexus.targ_port)]->fe_datamove;
 
 	fe_datamove(io);
 
 	return;
 
 }
 
 static int
 ctl_datamove_remote_dm_read_cb(union ctl_io *io)
 {
 #if 0
 	char str[256];
 	char path_str[64];
 	struct sbuf sb;
 #endif
 
 	/*
 	 * In this case, we had to malloc the memory locally.  Free it.
 	 */
 	if ((io->io_hdr.flags & CTL_FLAG_AUTO_MIRROR) == 0) {
 		int i;
 		for (i = 0; i < io->scsiio.kern_sg_entries; i++)
 			free(io->io_hdr.local_sglist[i].addr, M_CTL);
 	}
 
 #if 0
 	scsi_path_string(io, path_str, sizeof(path_str));
 	sbuf_new(&sb, str, sizeof(str), SBUF_FIXEDLEN);
 	sbuf_cat(&sb, path_str);
 	scsi_command_string(&io->scsiio, NULL, &sb);
 	sbuf_printf(&sb, "\n");
 	sbuf_cat(&sb, path_str);
 	sbuf_printf(&sb, "Tag: 0x%04x, type %d\n",
 		    io->scsiio.tag_num, io->scsiio.tag_type);
 	sbuf_cat(&sb, path_str);
 	sbuf_printf(&sb, "%s: flags %#x, status %#x\n", __func__,
 		    io->io_hdr.flags, io->io_hdr.status);
 	sbuf_finish(&sb);
 	printk("%s", sbuf_data(&sb));
 #endif
 
 
 	/*
 	 * The read is done, now we need to send status (good or bad) back
 	 * to the other side.
 	 */
 	ctl_send_datamove_done(io, /*have_lock*/ 0);
 
 	return (0);
 }
 
 static void
 ctl_datamove_remote_read_cb(struct ctl_ha_dt_req *rq)
 {
 	union ctl_io *io;
 	void (*fe_datamove)(union ctl_io *io);
 
 	io = rq->context;
 
 	if (rq->ret != CTL_HA_STATUS_SUCCESS) {
 		printf("%s: ISC DMA read failed with error %d", __func__,
 		       rq->ret);
 		ctl_set_internal_failure(&io->scsiio,
 					 /*sks_valid*/ 1,
 					 /*retry_count*/ rq->ret);
 	}
 
 	ctl_dt_req_free(rq);
 
 	/* Switch the pointer over so the FETD knows what to do */
 	io->scsiio.kern_data_ptr = (uint8_t *)io->io_hdr.local_sglist;
 
 	/*
 	 * Use a custom move done callback, since we need to send completion
 	 * back to the other controller, not to the backend on this side.
 	 */
 	io->scsiio.be_move_done = ctl_datamove_remote_dm_read_cb;
 
 	/* XXX KDM add checks like the ones in ctl_datamove? */
 
 	fe_datamove = control_softc->ctl_ports[ctl_port_idx(io->io_hdr.nexus.targ_port)]->fe_datamove;
 
 	fe_datamove(io);
 }
 
 static int
 ctl_datamove_remote_sgl_setup(union ctl_io *io)
 {
 	struct ctl_sg_entry *local_sglist, *remote_sglist;
 	struct ctl_sg_entry *local_dma_sglist, *remote_dma_sglist;
 	struct ctl_softc *softc;
 	int retval;
 	int i;
 
 	retval = 0;
 	softc = control_softc;
 
 	local_sglist = io->io_hdr.local_sglist;
 	local_dma_sglist = io->io_hdr.local_dma_sglist;
 	remote_sglist = io->io_hdr.remote_sglist;
 	remote_dma_sglist = io->io_hdr.remote_dma_sglist;
 
 	if (io->io_hdr.flags & CTL_FLAG_AUTO_MIRROR) {
 		for (i = 0; i < io->scsiio.kern_sg_entries; i++) {
 			local_sglist[i].len = remote_sglist[i].len;
 
 			/*
 			 * XXX Detect the situation where the RS-level I/O
 			 * redirector on the other side has already read the
 			 * data off of the AOR RS on this side, and
 			 * transferred it to remote (mirror) memory on the
 			 * other side.  Since we already have the data in
 			 * memory here, we just need to use it.
 			 *
 			 * XXX KDM this can probably be removed once we
 			 * get the cache device code in and take the
 			 * current AOR implementation out.
 			 */
 #ifdef NEEDTOPORT
 			if ((remote_sglist[i].addr >=
 			     (void *)vtophys(softc->mirr->addr))
 			 && (remote_sglist[i].addr <
 			     ((void *)vtophys(softc->mirr->addr) +
 			     CacheMirrorOffset))) {
 				local_sglist[i].addr = remote_sglist[i].addr -
 					CacheMirrorOffset;
 				if ((io->io_hdr.flags & CTL_FLAG_DATA_MASK) ==
 				     CTL_FLAG_DATA_IN)
 					io->io_hdr.flags |= CTL_FLAG_REDIR_DONE;
 			} else {
 				local_sglist[i].addr = remote_sglist[i].addr +
 					CacheMirrorOffset;
 			}
 #endif
 #if 0
 			printf("%s: local %p, remote %p, len %d\n",
 			       __func__, local_sglist[i].addr,
 			       remote_sglist[i].addr, local_sglist[i].len);
 #endif
 		}
 	} else {
 		uint32_t len_to_go;
 
 		/*
 		 * In this case, we don't have automatically allocated
 		 * memory for this I/O on this controller.  This typically
 		 * happens with internal CTL I/O -- e.g. inquiry, mode
 		 * sense, etc.  Anything coming from RAIDCore will have
 		 * a mirror area available.
 		 */
 		len_to_go = io->scsiio.kern_data_len;
 
 		/*
 		 * Clear the no datasync flag, we have to use malloced
 		 * buffers.
 		 */
 		io->io_hdr.flags &= ~CTL_FLAG_NO_DATASYNC;
 
 		/*
 		 * The difficult thing here is that the size of the various
 		 * S/G segments may be different than the size from the
 		 * remote controller.  That'll make it harder when DMAing
 		 * the data back to the other side.
 		 */
 		for (i = 0; (i < sizeof(io->io_hdr.remote_sglist) /
 		     sizeof(io->io_hdr.remote_sglist[0])) &&
 		     (len_to_go > 0); i++) {
 			local_sglist[i].len = MIN(len_to_go, 131072);
 			CTL_SIZE_8B(local_dma_sglist[i].len,
 				    local_sglist[i].len);
 			local_sglist[i].addr =
 				malloc(local_dma_sglist[i].len, M_CTL,M_WAITOK);
 
 			local_dma_sglist[i].addr = local_sglist[i].addr;
 
 			if (local_sglist[i].addr == NULL) {
 				int j;
 
 				printf("malloc failed for %zd bytes!",
 				       local_dma_sglist[i].len);
 				for (j = 0; j < i; j++) {
 					free(local_sglist[j].addr, M_CTL);
 				}
 				ctl_set_internal_failure(&io->scsiio,
 							 /*sks_valid*/ 1,
 							 /*retry_count*/ 4857);
 				retval = 1;
 				goto bailout_error;
 				
 			}
 			/* XXX KDM do we need a sync here? */
 
 			len_to_go -= local_sglist[i].len;
 		}
 		/*
 		 * Reset the number of S/G entries accordingly.  The
 		 * original number of S/G entries is available in
 		 * rem_sg_entries.
 		 */
 		io->scsiio.kern_sg_entries = i;
 
 #if 0
 		printf("%s: kern_sg_entries = %d\n", __func__,
 		       io->scsiio.kern_sg_entries);
 		for (i = 0; i < io->scsiio.kern_sg_entries; i++)
 			printf("%s: sg[%d] = %p, %d (DMA: %d)\n", __func__, i,
 			       local_sglist[i].addr, local_sglist[i].len,
 			       local_dma_sglist[i].len);
 #endif
 	}
 
 
 	return (retval);
 
 bailout_error:
 
 	ctl_send_datamove_done(io, /*have_lock*/ 0);
 
 	return (retval);
 }
 
 static int
 ctl_datamove_remote_xfer(union ctl_io *io, unsigned command,
 			 ctl_ha_dt_cb callback)
 {
 	struct ctl_ha_dt_req *rq;
 	struct ctl_sg_entry *remote_sglist, *local_sglist;
 	struct ctl_sg_entry *remote_dma_sglist, *local_dma_sglist;
 	uint32_t local_used, remote_used, total_used;
 	int retval;
 	int i, j;
 
 	retval = 0;
 
 	rq = ctl_dt_req_alloc();
 
 	/*
 	 * If we failed to allocate the request, and if the DMA didn't fail
 	 * anyway, set busy status.  This is just a resource allocation
 	 * failure.
 	 */
 	if ((rq == NULL)
 	 && ((io->io_hdr.status & CTL_STATUS_MASK) != CTL_STATUS_NONE))
 		ctl_set_busy(&io->scsiio);
 
 	if ((io->io_hdr.status & CTL_STATUS_MASK) != CTL_STATUS_NONE) {
 
 		if (rq != NULL)
 			ctl_dt_req_free(rq);
 
 		/*
 		 * The data move failed.  We need to return status back
 		 * to the other controller.  No point in trying to DMA
 		 * data to the remote controller.
 		 */
 
 		ctl_send_datamove_done(io, /*have_lock*/ 0);
 
 		retval = 1;
 
 		goto bailout;
 	}
 
 	local_sglist = io->io_hdr.local_sglist;
 	local_dma_sglist = io->io_hdr.local_dma_sglist;
 	remote_sglist = io->io_hdr.remote_sglist;
 	remote_dma_sglist = io->io_hdr.remote_dma_sglist;
 	local_used = 0;
 	remote_used = 0;
 	total_used = 0;
 
 	if (io->io_hdr.flags & CTL_FLAG_REDIR_DONE) {
 		rq->ret = CTL_HA_STATUS_SUCCESS;
 		rq->context = io;
 		callback(rq);
 		goto bailout;
 	}
 
 	/*
 	 * Pull/push the data over the wire from/to the other controller.
 	 * This takes into account the possibility that the local and
 	 * remote sglists may not be identical in terms of the size of
 	 * the elements and the number of elements.
 	 *
 	 * One fundamental assumption here is that the length allocated for
 	 * both the local and remote sglists is identical.  Otherwise, we've
 	 * essentially got a coding error of some sort.
 	 */
 	for (i = 0, j = 0; total_used < io->scsiio.kern_data_len; ) {
 		int isc_ret;
 		uint32_t cur_len, dma_length;
 		uint8_t *tmp_ptr;
 
 		rq->id = CTL_HA_DATA_CTL;
 		rq->command = command;
 		rq->context = io;
 
 		/*
 		 * Both pointers should be aligned.  But it is possible
 		 * that the allocation length is not.  They should both
 		 * also have enough slack left over at the end, though,
 		 * to round up to the next 8 byte boundary.
 		 */
 		cur_len = MIN(local_sglist[i].len - local_used,
 			      remote_sglist[j].len - remote_used);
 
 		/*
 		 * In this case, we have a size issue and need to decrease
 		 * the size, except in the case where we actually have less
 		 * than 8 bytes left.  In that case, we need to increase
 		 * the DMA length to get the last bit.
 		 */
 		if ((cur_len & 0x7) != 0) {
 			if (cur_len > 0x7) {
 				cur_len = cur_len - (cur_len & 0x7);
 				dma_length = cur_len;
 			} else {
 				CTL_SIZE_8B(dma_length, cur_len);
 			}
 
 		} else
 			dma_length = cur_len;
 
 		/*
 		 * If we had to allocate memory for this I/O, instead of using
 		 * the non-cached mirror memory, we'll need to flush the cache
 		 * before trying to DMA to the other controller.
 		 *
 		 * We could end up doing this multiple times for the same
 		 * segment if we have a larger local segment than remote
 		 * segment.  That shouldn't be an issue.
 		 */
 		if ((io->io_hdr.flags & CTL_FLAG_NO_DATASYNC) == 0) {
 			/*
 			 * XXX KDM use bus_dmamap_sync() here.
 			 */
 		}
 
 		rq->size = dma_length;
 
 		tmp_ptr = (uint8_t *)local_sglist[i].addr;
 		tmp_ptr += local_used;
 
 		/* Use physical addresses when talking to ISC hardware */
 		if ((io->io_hdr.flags & CTL_FLAG_BUS_ADDR) == 0) {
 			/* XXX KDM use busdma */
 #if 0
 			rq->local = vtophys(tmp_ptr);
 #endif
 		} else
 			rq->local = tmp_ptr;
 
 		tmp_ptr = (uint8_t *)remote_sglist[j].addr;
 		tmp_ptr += remote_used;
 		rq->remote = tmp_ptr;
 
 		rq->callback = NULL;
 
 		local_used += cur_len;
 		if (local_used >= local_sglist[i].len) {
 			i++;
 			local_used = 0;
 		}
 
 		remote_used += cur_len;
 		if (remote_used >= remote_sglist[j].len) {
 			j++;
 			remote_used = 0;
 		}
 		total_used += cur_len;
 
 		if (total_used >= io->scsiio.kern_data_len)
 			rq->callback = callback;
 
 		if ((rq->size & 0x7) != 0) {
 			printf("%s: warning: size %d is not on 8b boundary\n",
 			       __func__, rq->size);
 		}
 		if (((uintptr_t)rq->local & 0x7) != 0) {
 			printf("%s: warning: local %p not on 8b boundary\n",
 			       __func__, rq->local);
 		}
 		if (((uintptr_t)rq->remote & 0x7) != 0) {
 			printf("%s: warning: remote %p not on 8b boundary\n",
 			       __func__, rq->local);
 		}
 #if 0
 		printf("%s: %s: local %#x remote %#x size %d\n", __func__,
 		       (command == CTL_HA_DT_CMD_WRITE) ? "WRITE" : "READ",
 		       rq->local, rq->remote, rq->size);
 #endif
 
 		isc_ret = ctl_dt_single(rq);
 		if (isc_ret == CTL_HA_STATUS_WAIT)
 			continue;
 
 		if (isc_ret == CTL_HA_STATUS_DISCONNECT) {
 			rq->ret = CTL_HA_STATUS_SUCCESS;
 		} else {
 			rq->ret = isc_ret;
 		}
 		callback(rq);
 		goto bailout;
 	}
 
 bailout:
 	return (retval);
 
 }
 
 static void
 ctl_datamove_remote_read(union ctl_io *io)
 {
 	int retval;
 	int i;
 
 	/*
 	 * This will send an error to the other controller in the case of a
 	 * failure.
 	 */
 	retval = ctl_datamove_remote_sgl_setup(io);
 	if (retval != 0)
 		return;
 
 	retval = ctl_datamove_remote_xfer(io, CTL_HA_DT_CMD_READ,
 					  ctl_datamove_remote_read_cb);
 	if ((retval != 0)
 	 && ((io->io_hdr.flags & CTL_FLAG_AUTO_MIRROR) == 0)) {
 		/*
 		 * Make sure we free memory if there was an error..  The
 		 * ctl_datamove_remote_xfer() function will send the
 		 * datamove done message, or call the callback with an
 		 * error if there is a problem.
 		 */
 		for (i = 0; i < io->scsiio.kern_sg_entries; i++)
 			free(io->io_hdr.local_sglist[i].addr, M_CTL);
 	}
 
 	return;
 }
 
 /*
  * Process a datamove request from the other controller.  This is used for
  * XFER mode only, not SER_ONLY mode.  For writes, we DMA into local memory
  * first.  Once that is complete, the data gets DMAed into the remote
  * controller's memory.  For reads, we DMA from the remote controller's
  * memory into our memory first, and then move it out to the FETD.
  */
 static void
 ctl_datamove_remote(union ctl_io *io)
 {
 	struct ctl_softc *softc;
 
 	softc = control_softc;
 
 	mtx_assert(&softc->ctl_lock, MA_NOTOWNED);
 
 	/*
 	 * Note that we look for an aborted I/O here, but don't do some of
 	 * the other checks that ctl_datamove() normally does.
 	 * We don't need to run the datamove delay code, since that should
 	 * have been done if need be on the other controller.
 	 */
 	if (io->io_hdr.flags & CTL_FLAG_ABORT) {
 		printf("%s: tag 0x%04x on (%d:%d:%d:%d) aborted\n", __func__,
 		       io->scsiio.tag_num, io->io_hdr.nexus.initid.id,
 		       io->io_hdr.nexus.targ_port,
 		       io->io_hdr.nexus.targ_target.id,
 		       io->io_hdr.nexus.targ_lun);
 		io->io_hdr.port_status = 31338;
 		ctl_send_datamove_done(io, /*have_lock*/ 0);
 		return;
 	}
 
 	if ((io->io_hdr.flags & CTL_FLAG_DATA_MASK) == CTL_FLAG_DATA_OUT) {
 		ctl_datamove_remote_write(io);
 	} else if ((io->io_hdr.flags & CTL_FLAG_DATA_MASK) == CTL_FLAG_DATA_IN){
 		ctl_datamove_remote_read(io);
 	} else {
 		union ctl_ha_msg msg;
 		struct scsi_sense_data *sense;
 		uint8_t sks[3];
 		int retry_count;
 
 		memset(&msg, 0, sizeof(msg));
 
 		msg.hdr.msg_type = CTL_MSG_BAD_JUJU;
 		msg.hdr.status = CTL_SCSI_ERROR;
 		msg.scsi.scsi_status = SCSI_STATUS_CHECK_COND;
 
 		retry_count = 4243;
 
 		sense = &msg.scsi.sense_data;
 		sks[0] = SSD_SCS_VALID;
 		sks[1] = (retry_count >> 8) & 0xff;
 		sks[2] = retry_count & 0xff;
 
 		/* "Internal target failure" */
 		scsi_set_sense_data(sense,
 				    /*sense_format*/ SSD_TYPE_NONE,
 				    /*current_error*/ 1,
 				    /*sense_key*/ SSD_KEY_HARDWARE_ERROR,
 				    /*asc*/ 0x44,
 				    /*ascq*/ 0x00,
 				    /*type*/ SSD_ELEM_SKS,
 				    /*size*/ sizeof(sks),
 				    /*data*/ sks,
 				    SSD_ELEM_NONE);
 
 		io->io_hdr.flags &= ~CTL_FLAG_IO_ACTIVE;
 		if (io->io_hdr.flags & CTL_FLAG_FAILOVER) {
 			ctl_failover_io(io, /*have_lock*/ 1);
 			return;
 		}
 
 		if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg, sizeof(msg), 0) >
 		    CTL_HA_STATUS_SUCCESS) {
 			/* XXX KDM what to do if this fails? */
 		}
 		return;
 	}
 	
 }
 
 static int
 ctl_process_done(union ctl_io *io)
 {
 	struct ctl_lun *lun;
 	struct ctl_softc *softc = control_softc;
 	void (*fe_done)(union ctl_io *io);
 	uint32_t targ_port = ctl_port_idx(io->io_hdr.nexus.targ_port);
 
 	CTL_DEBUG_PRINT(("ctl_process_done\n"));
 
 	fe_done = softc->ctl_ports[targ_port]->fe_done;
 
 #ifdef CTL_TIME_IO
 	if ((time_uptime - io->io_hdr.start_time) > ctl_time_io_secs) {
 		char str[256];
 		char path_str[64];
 		struct sbuf sb;
 
 		ctl_scsi_path_string(io, path_str, sizeof(path_str));
 		sbuf_new(&sb, str, sizeof(str), SBUF_FIXEDLEN);
 
 		sbuf_cat(&sb, path_str);
 		switch (io->io_hdr.io_type) {
 		case CTL_IO_SCSI:
 			ctl_scsi_command_string(&io->scsiio, NULL, &sb);
 			sbuf_printf(&sb, "\n");
 			sbuf_cat(&sb, path_str);
 			sbuf_printf(&sb, "Tag: 0x%04x, type %d\n",
 				    io->scsiio.tag_num, io->scsiio.tag_type);
 			break;
 		case CTL_IO_TASK:
 			sbuf_printf(&sb, "Task I/O type: %d, Tag: 0x%04x, "
 				    "Tag Type: %d\n", io->taskio.task_action,
 				    io->taskio.tag_num, io->taskio.tag_type);
 			break;
 		default:
 			printf("Invalid CTL I/O type %d\n", io->io_hdr.io_type);
 			panic("Invalid CTL I/O type %d\n", io->io_hdr.io_type);
 			break;
 		}
 		sbuf_cat(&sb, path_str);
 		sbuf_printf(&sb, "ctl_process_done: %jd seconds\n",
 			    (intmax_t)time_uptime - io->io_hdr.start_time);
 		sbuf_finish(&sb);
 		printf("%s", sbuf_data(&sb));
 	}
 #endif /* CTL_TIME_IO */
 
 	switch (io->io_hdr.io_type) {
 	case CTL_IO_SCSI:
 		break;
 	case CTL_IO_TASK:
 		if (bootverbose || (ctl_debug & CTL_DEBUG_INFO))
 			ctl_io_error_print(io, NULL);
 		if (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC)
 			ctl_free_io(io);
 		else
 			fe_done(io);
 		return (CTL_RETVAL_COMPLETE);
 	default:
 		panic("ctl_process_done: invalid io type %d\n",
 		      io->io_hdr.io_type);
 		break; /* NOTREACHED */
 	}
 
 	lun = (struct ctl_lun *)io->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 	if (lun == NULL) {
 		CTL_DEBUG_PRINT(("NULL LUN for lun %d\n",
 				 io->io_hdr.nexus.targ_mapped_lun));
 		goto bailout;
 	}
 
 	mtx_lock(&lun->lun_lock);
 
 	/*
 	 * Check to see if we have any errors to inject here.  We only
 	 * inject errors for commands that don't already have errors set.
 	 */
 	if ((STAILQ_FIRST(&lun->error_list) != NULL) &&
 	    ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_SUCCESS) &&
 	    ((io->io_hdr.flags & CTL_FLAG_STATUS_SENT) == 0))
 		ctl_inject_error(lun, io);
 
 	/*
 	 * XXX KDM how do we treat commands that aren't completed
 	 * successfully?
 	 *
 	 * XXX KDM should we also track I/O latency?
 	 */
 	if ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_SUCCESS &&
 	    io->io_hdr.io_type == CTL_IO_SCSI) {
 #ifdef CTL_TIME_IO
 		struct bintime cur_bt;
 #endif
 		int type;
 
 		if ((io->io_hdr.flags & CTL_FLAG_DATA_MASK) ==
 		    CTL_FLAG_DATA_IN)
 			type = CTL_STATS_READ;
 		else if ((io->io_hdr.flags & CTL_FLAG_DATA_MASK) ==
 		    CTL_FLAG_DATA_OUT)
 			type = CTL_STATS_WRITE;
 		else
 			type = CTL_STATS_NO_IO;
 
 		lun->stats.ports[targ_port].bytes[type] +=
 		    io->scsiio.kern_total_len;
 		lun->stats.ports[targ_port].operations[type]++;
 #ifdef CTL_TIME_IO
 		bintime_add(&lun->stats.ports[targ_port].dma_time[type],
 		   &io->io_hdr.dma_bt);
 		lun->stats.ports[targ_port].num_dmas[type] +=
 		    io->io_hdr.num_dmas;
 		getbintime(&cur_bt);
 		bintime_sub(&cur_bt, &io->io_hdr.start_bt);
 		bintime_add(&lun->stats.ports[targ_port].time[type], &cur_bt);
 #endif
 	}
 
 	/*
 	 * Remove this from the OOA queue.
 	 */
 	TAILQ_REMOVE(&lun->ooa_queue, &io->io_hdr, ooa_links);
 
 	/*
 	 * Run through the blocked queue on this LUN and see if anything
 	 * has become unblocked, now that this transaction is done.
 	 */
 	ctl_check_blocked(lun);
 
 	/*
 	 * If the LUN has been invalidated, free it if there is nothing
 	 * left on its OOA queue.
 	 */
 	if ((lun->flags & CTL_LUN_INVALID)
 	 && TAILQ_EMPTY(&lun->ooa_queue)) {
 		mtx_unlock(&lun->lun_lock);
 		mtx_lock(&softc->ctl_lock);
 		ctl_free_lun(lun);
 		mtx_unlock(&softc->ctl_lock);
 	} else
 		mtx_unlock(&lun->lun_lock);
 
 bailout:
 
 	/*
 	 * If this command has been aborted, make sure we set the status
 	 * properly.  The FETD is responsible for freeing the I/O and doing
 	 * whatever it needs to do to clean up its state.
 	 */
 	if (io->io_hdr.flags & CTL_FLAG_ABORT)
 		ctl_set_task_aborted(&io->scsiio);
 
 	/*
 	 * If enabled, print command error status.
 	 * We don't print UAs unless debugging was enabled explicitly.
 	 */
 	do {
 		if ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_SUCCESS)
 			break;
 		if (!bootverbose && (ctl_debug & CTL_DEBUG_INFO) == 0)
 			break;
 		if ((ctl_debug & CTL_DEBUG_INFO) == 0 &&
 		    ((io->io_hdr.status & CTL_STATUS_MASK) == CTL_SCSI_ERROR) &&
 		     (io->scsiio.scsi_status == SCSI_STATUS_CHECK_COND)) {
 			int error_code, sense_key, asc, ascq;
 
 			scsi_extract_sense_len(&io->scsiio.sense_data,
 			    io->scsiio.sense_len, &error_code, &sense_key,
 			    &asc, &ascq, /*show_errors*/ 0);
 			if (sense_key == SSD_KEY_UNIT_ATTENTION)
 				break;
 		}
 
 		ctl_io_error_print(io, NULL);
 	} while (0);
 
 	/*
 	 * Tell the FETD or the other shelf controller we're done with this
 	 * command.  Note that only SCSI commands get to this point.  Task
 	 * management commands are completed above.
 	 *
 	 * We only send status to the other controller if we're in XFER
 	 * mode.  In SER_ONLY mode, the I/O is done on the controller that
 	 * received the I/O (from CTL's perspective), and so the status is
 	 * generated there.
 	 * 
 	 * XXX KDM if we hold the lock here, we could cause a deadlock
 	 * if the frontend comes back in in this context to queue
 	 * something.
 	 */
 	if ((softc->ha_mode == CTL_HA_MODE_XFER)
 	 && (io->io_hdr.flags & CTL_FLAG_FROM_OTHER_SC)) {
 		union ctl_ha_msg msg;
 
 		memset(&msg, 0, sizeof(msg));
 		msg.hdr.msg_type = CTL_MSG_FINISH_IO;
 		msg.hdr.original_sc = io->io_hdr.original_sc;
 		msg.hdr.nexus = io->io_hdr.nexus;
 		msg.hdr.status = io->io_hdr.status;
 		msg.scsi.scsi_status = io->scsiio.scsi_status;
 		msg.scsi.tag_num = io->scsiio.tag_num;
 		msg.scsi.tag_type = io->scsiio.tag_type;
 		msg.scsi.sense_len = io->scsiio.sense_len;
 		msg.scsi.sense_residual = io->scsiio.sense_residual;
 		msg.scsi.residual = io->scsiio.residual;
 		memcpy(&msg.scsi.sense_data, &io->scsiio.sense_data,
 		       sizeof(io->scsiio.sense_data));
 		/*
 		 * We copy this whether or not this is an I/O-related
 		 * command.  Otherwise, we'd have to go and check to see
 		 * whether it's a read/write command, and it really isn't
 		 * worth it.
 		 */
 		memcpy(&msg.scsi.lbalen,
 		       &io->io_hdr.ctl_private[CTL_PRIV_LBA_LEN].bytes,
 		       sizeof(msg.scsi.lbalen));
 
 		if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg,
 				sizeof(msg), 0) > CTL_HA_STATUS_SUCCESS) {
 			/* XXX do something here */
 		}
 
 		ctl_free_io(io);
 	} else 
 		fe_done(io);
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 #ifdef CTL_WITH_CA
 /*
  * Front end should call this if it doesn't do autosense.  When the request
  * sense comes back in from the initiator, we'll dequeue this and send it.
  */
 int
 ctl_queue_sense(union ctl_io *io)
 {
 	struct ctl_lun *lun;
 	struct ctl_softc *softc;
 	uint32_t initidx, targ_lun;
 
 	softc = control_softc;
 
 	CTL_DEBUG_PRINT(("ctl_queue_sense\n"));
 
 	/*
 	 * LUN lookup will likely move to the ctl_work_thread() once we
 	 * have our new queueing infrastructure (that doesn't put things on
 	 * a per-LUN queue initially).  That is so that we can handle
 	 * things like an INQUIRY to a LUN that we don't have enabled.  We
 	 * can't deal with that right now.
 	 */
 	mtx_lock(&softc->ctl_lock);
 
 	/*
 	 * If we don't have a LUN for this, just toss the sense
 	 * information.
 	 */
 	targ_lun = io->io_hdr.nexus.targ_lun;
 	targ_lun = ctl_map_lun(softc, io->io_hdr.nexus.targ_port, targ_lun);
 	if ((targ_lun < CTL_MAX_LUNS)
 	 && (softc->ctl_luns[targ_lun] != NULL))
 		lun = softc->ctl_luns[targ_lun];
 	else
 		goto bailout;
 
 	initidx = ctl_get_initindex(&io->io_hdr.nexus);
 
 	mtx_lock(&lun->lun_lock);
 	/*
 	 * Already have CA set for this LUN...toss the sense information.
 	 */
 	if (ctl_is_set(lun->have_ca, initidx)) {
 		mtx_unlock(&lun->lun_lock);
 		goto bailout;
 	}
 
 	memcpy(&lun->pending_sense[initidx], &io->scsiio.sense_data,
 	       MIN(sizeof(lun->pending_sense[initidx]),
 	       sizeof(io->scsiio.sense_data)));
 	ctl_set_mask(lun->have_ca, initidx);
 	mtx_unlock(&lun->lun_lock);
 
 bailout:
 	mtx_unlock(&softc->ctl_lock);
 
 	ctl_free_io(io);
 
 	return (CTL_RETVAL_COMPLETE);
 }
 #endif
 
 /*
  * Primary command inlet from frontend ports.  All SCSI and task I/O
  * requests must go through this function.
  */
 int
 ctl_queue(union ctl_io *io)
 {
 
 	CTL_DEBUG_PRINT(("ctl_queue cdb[0]=%02X\n", io->scsiio.cdb[0]));
 
 #ifdef CTL_TIME_IO
 	io->io_hdr.start_time = time_uptime;
 	getbintime(&io->io_hdr.start_bt);
 #endif /* CTL_TIME_IO */
 
 	/* Map FE-specific LUN ID into global one. */
 	io->io_hdr.nexus.targ_mapped_lun =
 	    ctl_map_lun(control_softc, io->io_hdr.nexus.targ_port,
 	     io->io_hdr.nexus.targ_lun);
 
 	switch (io->io_hdr.io_type) {
 	case CTL_IO_SCSI:
 	case CTL_IO_TASK:
 		if (ctl_debug & CTL_DEBUG_CDB)
 			ctl_io_print(io);
 		ctl_enqueue_incoming(io);
 		break;
 	default:
 		printf("ctl_queue: unknown I/O type %d\n", io->io_hdr.io_type);
 		return (EINVAL);
 	}
 
 	return (CTL_RETVAL_COMPLETE);
 }
 
 #ifdef CTL_IO_DELAY
 static void
 ctl_done_timer_wakeup(void *arg)
 {
 	union ctl_io *io;
 
 	io = (union ctl_io *)arg;
 	ctl_done(io);
 }
 #endif /* CTL_IO_DELAY */
 
 void
 ctl_done(union ctl_io *io)
 {
 
 	/*
 	 * Enable this to catch duplicate completion issues.
 	 */
 #if 0
 	if (io->io_hdr.flags & CTL_FLAG_ALREADY_DONE) {
 		printf("%s: type %d msg %d cdb %x iptl: "
 		       "%d:%d:%d:%d tag 0x%04x "
 		       "flag %#x status %x\n",
 			__func__,
 			io->io_hdr.io_type,
 			io->io_hdr.msg_type,
 			io->scsiio.cdb[0],
 			io->io_hdr.nexus.initid.id,
 			io->io_hdr.nexus.targ_port,
 			io->io_hdr.nexus.targ_target.id,
 			io->io_hdr.nexus.targ_lun,
 			(io->io_hdr.io_type ==
 			CTL_IO_TASK) ?
 			io->taskio.tag_num :
 			io->scsiio.tag_num,
 		        io->io_hdr.flags,
 			io->io_hdr.status);
 	} else
 		io->io_hdr.flags |= CTL_FLAG_ALREADY_DONE;
 #endif
 
 	/*
 	 * This is an internal copy of an I/O, and should not go through
 	 * the normal done processing logic.
 	 */
 	if (io->io_hdr.flags & CTL_FLAG_INT_COPY)
 		return;
 
 	/*
 	 * We need to send a msg to the serializing shelf to finish the IO
 	 * as well.  We don't send a finish message to the other shelf if
 	 * this is a task management command.  Task management commands
 	 * aren't serialized in the OOA queue, but rather just executed on
 	 * both shelf controllers for commands that originated on that
 	 * controller.
 	 */
 	if ((io->io_hdr.flags & CTL_FLAG_SENT_2OTHER_SC)
 	 && (io->io_hdr.io_type != CTL_IO_TASK)) {
 		union ctl_ha_msg msg_io;
 
 		msg_io.hdr.msg_type = CTL_MSG_FINISH_IO;
 		msg_io.hdr.serializing_sc = io->io_hdr.serializing_sc;
 		if (ctl_ha_msg_send(CTL_HA_CHAN_CTL, &msg_io,
 		    sizeof(msg_io), 0 ) != CTL_HA_STATUS_SUCCESS) {
 		}
 		/* continue on to finish IO */
 	}
 #ifdef CTL_IO_DELAY
 	if (io->io_hdr.flags & CTL_FLAG_DELAY_DONE) {
 		struct ctl_lun *lun;
 
 		lun =(struct ctl_lun *)io->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 		io->io_hdr.flags &= ~CTL_FLAG_DELAY_DONE;
 	} else {
 		struct ctl_lun *lun;
 
 		lun =(struct ctl_lun *)io->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 		if ((lun != NULL)
 		 && (lun->delay_info.done_delay > 0)) {
 			struct callout *callout;
 
 			callout = (struct callout *)&io->io_hdr.timer_bytes;
 			callout_init(callout, /*mpsafe*/ 1);
 			io->io_hdr.flags |= CTL_FLAG_DELAY_DONE;
 			callout_reset(callout,
 				      lun->delay_info.done_delay * hz,
 				      ctl_done_timer_wakeup, io);
 			if (lun->delay_info.done_type == CTL_DELAY_TYPE_ONESHOT)
 				lun->delay_info.done_delay = 0;
 			return;
 		}
 	}
 #endif /* CTL_IO_DELAY */
 
 	ctl_enqueue_done(io);
 }
 
 int
 ctl_isc(struct ctl_scsiio *ctsio)
 {
 	struct ctl_lun *lun;
 	int retval;
 
 	lun = (struct ctl_lun *)ctsio->io_hdr.ctl_private[CTL_PRIV_LUN].ptr;
 
 	CTL_DEBUG_PRINT(("ctl_isc: command: %02x\n", ctsio->cdb[0]));
 
 	CTL_DEBUG_PRINT(("ctl_isc: calling data_submit()\n"));
 
 	retval = lun->backend->data_submit((union ctl_io *)ctsio);
 
 	return (retval);
 }
 
 
 static void
 ctl_work_thread(void *arg)
 {
 	struct ctl_thread *thr = (struct ctl_thread *)arg;
 	struct ctl_softc *softc = thr->ctl_softc;
 	union ctl_io *io;
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_work_thread starting\n"));
 
 	for (;;) {
 		retval = 0;
 
 		/*
 		 * We handle the queues in this order:
 		 * - ISC
 		 * - done queue (to free up resources, unblock other commands)
 		 * - RtR queue
 		 * - incoming queue
 		 *
 		 * If those queues are empty, we break out of the loop and
 		 * go to sleep.
 		 */
 		mtx_lock(&thr->queue_lock);
 		io = (union ctl_io *)STAILQ_FIRST(&thr->isc_queue);
 		if (io != NULL) {
 			STAILQ_REMOVE_HEAD(&thr->isc_queue, links);
 			mtx_unlock(&thr->queue_lock);
 			ctl_handle_isc(io);
 			continue;
 		}
 		io = (union ctl_io *)STAILQ_FIRST(&thr->done_queue);
 		if (io != NULL) {
 			STAILQ_REMOVE_HEAD(&thr->done_queue, links);
 			/* clear any blocked commands, call fe_done */
 			mtx_unlock(&thr->queue_lock);
 			retval = ctl_process_done(io);
 			continue;
 		}
 		io = (union ctl_io *)STAILQ_FIRST(&thr->incoming_queue);
 		if (io != NULL) {
 			STAILQ_REMOVE_HEAD(&thr->incoming_queue, links);
 			mtx_unlock(&thr->queue_lock);
 			if (io->io_hdr.io_type == CTL_IO_TASK)
 				ctl_run_task(io);
 			else
 				ctl_scsiio_precheck(softc, &io->scsiio);
 			continue;
 		}
 		if (!ctl_pause_rtr) {
 			io = (union ctl_io *)STAILQ_FIRST(&thr->rtr_queue);
 			if (io != NULL) {
 				STAILQ_REMOVE_HEAD(&thr->rtr_queue, links);
 				mtx_unlock(&thr->queue_lock);
 				retval = ctl_scsiio(&io->scsiio);
 				if (retval != CTL_RETVAL_COMPLETE)
 					CTL_DEBUG_PRINT(("ctl_scsiio failed\n"));
 				continue;
 			}
 		}
 
 		/* Sleep until we have something to do. */
 		mtx_sleep(thr, &thr->queue_lock, PDROP | PRIBIO, "-", 0);
 	}
 }
 
 static void
 ctl_lun_thread(void *arg)
 {
 	struct ctl_softc *softc = (struct ctl_softc *)arg;
 	struct ctl_be_lun *be_lun;
 	int retval;
 
 	CTL_DEBUG_PRINT(("ctl_lun_thread starting\n"));
 
 	for (;;) {
 		retval = 0;
 		mtx_lock(&softc->ctl_lock);
 		be_lun = STAILQ_FIRST(&softc->pending_lun_queue);
 		if (be_lun != NULL) {
 			STAILQ_REMOVE_HEAD(&softc->pending_lun_queue, links);
 			mtx_unlock(&softc->ctl_lock);
 			ctl_create_lun(be_lun);
 			continue;
 		}
 
 		/* Sleep until we have something to do. */
 		mtx_sleep(&softc->pending_lun_queue, &softc->ctl_lock,
 		    PDROP | PRIBIO, "-", 0);
 	}
 }
 
 static void
 ctl_thresh_thread(void *arg)
 {
 	struct ctl_softc *softc = (struct ctl_softc *)arg;
 	struct ctl_lun *lun;
 	struct ctl_be_lun *be_lun;
 	struct scsi_da_rw_recovery_page *rwpage;
 	struct ctl_logical_block_provisioning_page *page;
 	const char *attr;
 	uint64_t thres, val;
 	int i, e;
 
 	CTL_DEBUG_PRINT(("ctl_thresh_thread starting\n"));
 
 	for (;;) {
 		mtx_lock(&softc->ctl_lock);
 		STAILQ_FOREACH(lun, &softc->lun_list, links) {
 			be_lun = lun->be_lun;
 			if ((lun->flags & CTL_LUN_DISABLED) ||
 			    (lun->flags & CTL_LUN_OFFLINE) ||
 			    lun->backend->lun_attr == NULL)
 				continue;
 			rwpage = &lun->mode_pages.rw_er_page[CTL_PAGE_CURRENT];
 			if ((rwpage->byte8 & SMS_RWER_LBPERE) == 0)
 				continue;
 			e = 0;
 			page = &lun->mode_pages.lbp_page[CTL_PAGE_CURRENT];
 			for (i = 0; i < CTL_NUM_LBP_THRESH; i++) {
 				if ((page->descr[i].flags & SLBPPD_ENABLED) == 0)
 					continue;
 				thres = scsi_4btoul(page->descr[i].count);
 				thres <<= CTL_LBP_EXPONENT;
 				switch (page->descr[i].resource) {
 				case 0x01:
 					attr = "blocksavail";
 					break;
 				case 0x02:
 					attr = "blocksused";
 					break;
 				case 0xf1:
 					attr = "poolblocksavail";
 					break;
 				case 0xf2:
 					attr = "poolblocksused";
 					break;
 				default:
 					continue;
 				}
 				mtx_unlock(&softc->ctl_lock); // XXX
 				val = lun->backend->lun_attr(
 				    lun->be_lun->be_lun, attr);
 				mtx_lock(&softc->ctl_lock);
 				if (val == UINT64_MAX)
 					continue;
 				if ((page->descr[i].flags & SLBPPD_ARMING_MASK)
 				    == SLBPPD_ARMING_INC)
 					e |= (val >= thres);
 				else
 					e |= (val <= thres);
 			}
 			mtx_lock(&lun->lun_lock);
 			if (e) {
 				if (lun->lasttpt == 0 ||
 				    time_uptime - lun->lasttpt >= CTL_LBP_UA_PERIOD) {
 					lun->lasttpt = time_uptime;
 					ctl_est_ua_all(lun, -1, CTL_UA_THIN_PROV_THRES);
 				}
 			} else {
 				lun->lasttpt = 0;
 				ctl_clr_ua_all(lun, -1, CTL_UA_THIN_PROV_THRES);
 			}
 			mtx_unlock(&lun->lun_lock);
 		}
 		mtx_unlock(&softc->ctl_lock);
 		pause("-", CTL_LBP_PERIOD * hz);
 	}
 }
 
 static void
 ctl_enqueue_incoming(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_thread *thr;
 	u_int idx;
 
 	idx = (io->io_hdr.nexus.targ_port * 127 +
 	       io->io_hdr.nexus.initid.id) % worker_threads;
 	thr = &softc->threads[idx];
 	mtx_lock(&thr->queue_lock);
 	STAILQ_INSERT_TAIL(&thr->incoming_queue, &io->io_hdr, links);
 	mtx_unlock(&thr->queue_lock);
 	wakeup(thr);
 }
 
 static void
 ctl_enqueue_rtr(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_thread *thr;
 
 	thr = &softc->threads[io->io_hdr.nexus.targ_mapped_lun % worker_threads];
 	mtx_lock(&thr->queue_lock);
 	STAILQ_INSERT_TAIL(&thr->rtr_queue, &io->io_hdr, links);
 	mtx_unlock(&thr->queue_lock);
 	wakeup(thr);
 }
 
 static void
 ctl_enqueue_done(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_thread *thr;
 
 	thr = &softc->threads[io->io_hdr.nexus.targ_mapped_lun % worker_threads];
 	mtx_lock(&thr->queue_lock);
 	STAILQ_INSERT_TAIL(&thr->done_queue, &io->io_hdr, links);
 	mtx_unlock(&thr->queue_lock);
 	wakeup(thr);
 }
 
 static void
 ctl_enqueue_isc(union ctl_io *io)
 {
 	struct ctl_softc *softc = control_softc;
 	struct ctl_thread *thr;
 
 	thr = &softc->threads[io->io_hdr.nexus.targ_mapped_lun % worker_threads];
 	mtx_lock(&thr->queue_lock);
 	STAILQ_INSERT_TAIL(&thr->isc_queue, &io->io_hdr, links);
 	mtx_unlock(&thr->queue_lock);
 	wakeup(thr);
 }
 
 /* Initialization and failover */
 
 void
 ctl_init_isc_msg(void)
 {
 	printf("CTL: Still calling this thing\n");
 }
 
 /*
  * Init component
  * 	Initializes component into configuration defined by bootMode
  *	(see hasc-sv.c)
  *  	returns hasc_Status:
  * 		OK
  *		ERROR - fatal error
  */
 static ctl_ha_comp_status
 ctl_isc_init(struct ctl_ha_component *c)
 {
 	ctl_ha_comp_status ret = CTL_HA_COMP_STATUS_OK;
 
 	c->status = ret;
 	return ret;
 }
 
 /* Start component
  * 	Starts component in state requested. If component starts successfully,
  *	it must set its own state to the requestrd state
  *	When requested state is HASC_STATE_HA, the component may refine it
  * 	by adding _SLAVE or _MASTER flags.
  *	Currently allowed state transitions are:
  *	UNKNOWN->HA		- initial startup
  *	UNKNOWN->SINGLE - initial startup when no parter detected
  *	HA->SINGLE		- failover
  * returns ctl_ha_comp_status:
  * 		OK	- component successfully started in requested state
  *		FAILED  - could not start the requested state, failover may
  * 			  be possible
  *		ERROR	- fatal error detected, no future startup possible
  */
 static ctl_ha_comp_status
 ctl_isc_start(struct ctl_ha_component *c, ctl_ha_state state)
 {
 	ctl_ha_comp_status ret = CTL_HA_COMP_STATUS_OK;
 
 	printf("%s: go\n", __func__);
 
 	// UNKNOWN->HA or UNKNOWN->SINGLE (bootstrap)
 	if (c->state == CTL_HA_STATE_UNKNOWN ) {
 		control_softc->is_single = 0;
 		if (ctl_ha_msg_create(CTL_HA_CHAN_CTL, ctl_isc_event_handler)
 		    != CTL_HA_STATUS_SUCCESS) {
 			printf("ctl_isc_start: ctl_ha_msg_create failed.\n");
 			ret = CTL_HA_COMP_STATUS_ERROR;
 		}
 	} else if (CTL_HA_STATE_IS_HA(c->state)
 		&& CTL_HA_STATE_IS_SINGLE(state)){
 		// HA->SINGLE transition
 	        ctl_failover();
 		control_softc->is_single = 1;
 	} else {
 		printf("ctl_isc_start:Invalid state transition %X->%X\n",
 		       c->state, state);
 		ret = CTL_HA_COMP_STATUS_ERROR;
 	}
 	if (CTL_HA_STATE_IS_SINGLE(state))
 		control_softc->is_single = 1;
 
 	c->state = state;
 	c->status = ret;
 	return ret;
 }
 
 /*
  * Quiesce component
  * The component must clear any error conditions (set status to OK) and
  * prepare itself to another Start call
  * returns ctl_ha_comp_status:
  * 	OK
  *	ERROR
  */
 static ctl_ha_comp_status
 ctl_isc_quiesce(struct ctl_ha_component *c)
 {
 	int ret = CTL_HA_COMP_STATUS_OK;
 
 	ctl_pause_rtr = 1;
 	c->status = ret;
 	return ret;
 }
 
 struct ctl_ha_component ctl_ha_component_ctlisc =
 {
 	.name = "CTL ISC",
 	.state = CTL_HA_STATE_UNKNOWN,
 	.init = ctl_isc_init,
 	.start = ctl_isc_start,
 	.quiesce = ctl_isc_quiesce
 };
 
 /*
  *  vim: ts=8
  */
Index: projects/clang360-import/sys/cam/scsi/scsi_all.h
===================================================================
--- projects/clang360-import/sys/cam/scsi/scsi_all.h	(revision 277944)
+++ projects/clang360-import/sys/cam/scsi/scsi_all.h	(revision 277945)
@@ -1,3746 +1,3760 @@
 /*-
  * Largely written by Julian Elischer (julian@tfs.com)
  * for TRW Financial Systems.
  *
  * TRW Financial Systems, in accordance with their agreement with Carnegie
  * Mellon University, makes this software available to CMU to distribute
  * or use in any manner that they see fit as long as this message is kept with
  * the software. For this reason TFS also grants any other persons or
  * organisations permission to use or modify this software.
  *
  * TFS supplies this software to be publicly redistributed
  * on the understanding that TFS is not responsible for the correct
  * functioning of this software in any circumstances.
  *
  * Ported to run under 386BSD by Julian Elischer (julian@tfs.com) Sept 1992
  *
  * $FreeBSD$
  */
 
 /*
  * SCSI general  interface description
  */
 
 #ifndef	_SCSI_SCSI_ALL_H
 #define	_SCSI_SCSI_ALL_H 1
 
 #include <sys/cdefs.h>
 #include <machine/stdarg.h>
 
 #ifdef _KERNEL
 /*
  * This is the number of seconds we wait for devices to settle after a SCSI
  * bus reset.
  */
 extern int scsi_delay;
 #endif /* _KERNEL */
 
 /*
  * SCSI command format
  */
 
 /*
  * Define dome bits that are in ALL (or a lot of) scsi commands
  */
 #define	SCSI_CTL_LINK		0x01
 #define	SCSI_CTL_FLAG		0x02
 #define	SCSI_CTL_VENDOR		0xC0
 #define	SCSI_CMD_LUN		0xA0	/* these two should not be needed */
 #define	SCSI_CMD_LUN_SHIFT	5	/* LUN in the cmd is no longer SCSI */
 
 #define	SCSI_MAX_CDBLEN		16	/* 
 					 * 16 byte commands are in the 
 					 * SCSI-3 spec 
 					 */
 #if defined(CAM_MAX_CDBLEN) && (CAM_MAX_CDBLEN < SCSI_MAX_CDBLEN)
 #error "CAM_MAX_CDBLEN cannot be less than SCSI_MAX_CDBLEN"
 #endif
 
 /* 6byte CDBs special case 0 length to be 256 */
 #define	SCSI_CDB6_LEN(len)	((len) == 0 ? 256 : len)
 
 /*
  * This type defines actions to be taken when a particular sense code is
  * received.  Right now, these flags are only defined to take up 16 bits,
  * but can be expanded in the future if necessary.
  */
 typedef enum {
 	SS_NOP      = 0x000000,	/* Do nothing */
 	SS_RETRY    = 0x010000,	/* Retry the command */
 	SS_FAIL     = 0x020000,	/* Bail out */
 	SS_START    = 0x030000,	/* Send a Start Unit command to the device,
 				 * then retry the original command.
 				 */
 	SS_TUR      = 0x040000,	/* Send a Test Unit Ready command to the
 				 * device, then retry the original command.
 				 */
 	SS_MASK     = 0xff0000
 } scsi_sense_action;
 
 typedef enum {
 	SSQ_NONE		= 0x0000,
 	SSQ_DECREMENT_COUNT	= 0x0100,  /* Decrement the retry count */
 	SSQ_MANY		= 0x0200,  /* send lots of recovery commands */
 	SSQ_RANGE		= 0x0400,  /*
 					    * This table entry represents the
 					    * end of a range of ASCQs that
 					    * have identical error actions
 					    * and text.
 					    */
 	SSQ_PRINT_SENSE		= 0x0800,
 	SSQ_UA			= 0x1000,  /* Broadcast UA. */
 	SSQ_RESCAN		= 0x2000,  /* Rescan target for LUNs. */
 	SSQ_LOST		= 0x4000,  /* Destroy the LUNs. */
 	SSQ_MASK		= 0xff00
 } scsi_sense_action_qualifier;
 
 /* Mask for error status values */
 #define	SS_ERRMASK	0xff
 
 /* The default, retyable, error action */
 #define	SS_RDEF		SS_RETRY|SSQ_DECREMENT_COUNT|SSQ_PRINT_SENSE|EIO
 
 /* The retyable, error action, with table specified error code */
 #define	SS_RET		SS_RETRY|SSQ_DECREMENT_COUNT|SSQ_PRINT_SENSE
 
 /* Fatal error action, with table specified error code */
 #define	SS_FATAL	SS_FAIL|SSQ_PRINT_SENSE
 
 struct scsi_generic
 {
 	u_int8_t opcode;
 	u_int8_t bytes[11];
 };
 
 struct scsi_request_sense
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SRS_DESC	0x01
 	u_int8_t unused[2];
 	u_int8_t length;
 	u_int8_t control;
 };
 
 struct scsi_test_unit_ready
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 	u_int8_t unused[3];
 	u_int8_t control;
 };
 
 struct scsi_receive_diag {
 	uint8_t opcode;
 	uint8_t byte2;
 #define SRD_PCV		0x01
 	uint8_t page_code;
 	uint8_t length[2]; 
 	uint8_t control;
 };
 
 struct scsi_send_diag {
 	uint8_t opcode;
 	uint8_t byte2;
 #define SSD_UNITOFFL				0x01
 #define SSD_DEVOFFL				0x02
 #define SSD_SELFTEST				0x04
 #define SSD_PF					0x10
 #define SSD_SELF_TEST_CODE_MASK			0xE0
 #define SSD_SELF_TEST_CODE_SHIFT		5
 #define		SSD_SELF_TEST_CODE_NONE		0x00
 #define		SSD_SELF_TEST_CODE_BG_SHORT	0x01
 #define		SSD_SELF_TEST_CODE_BG_EXTENDED	0x02
 #define		SSD_SELF_TEST_CODE_BG_ABORT	0x04
 #define		SSD_SELF_TEST_CODE_FG_SHORT	0x05
 #define		SSD_SELF_TEST_CODE_FG_EXTENDED	0x06
 	uint8_t	reserved;
 	uint8_t	length[2];
 	uint8_t control;
 };
 
 struct scsi_sense
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 	u_int8_t unused[2];
 	u_int8_t length;
 	u_int8_t control;
 };
 
 struct scsi_inquiry
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SI_EVPD 	0x01
 #define	SI_CMDDT	0x02
 	u_int8_t page_code;
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 struct scsi_mode_sense_6
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SMS_DBD				0x08
 	u_int8_t page;
 #define	SMS_PAGE_CODE 			0x3F
 #define	SMS_VENDOR_SPECIFIC_PAGE	0x00
 #define	SMS_DISCONNECT_RECONNECT_PAGE	0x02
 #define	SMS_FORMAT_DEVICE_PAGE		0x03
 #define	SMS_GEOMETRY_PAGE		0x04
 #define	SMS_CACHE_PAGE			0x08
 #define	SMS_PERIPHERAL_DEVICE_PAGE	0x09
 #define	SMS_CONTROL_MODE_PAGE		0x0A
 #define	SMS_PROTO_SPECIFIC_PAGE		0x19
 #define	SMS_INFO_EXCEPTIONS_PAGE	0x1C
 #define	SMS_ALL_PAGES_PAGE		0x3F
 #define	SMS_PAGE_CTRL_MASK		0xC0
 #define	SMS_PAGE_CTRL_CURRENT 		0x00
 #define	SMS_PAGE_CTRL_CHANGEABLE 	0x40
 #define	SMS_PAGE_CTRL_DEFAULT 		0x80
 #define	SMS_PAGE_CTRL_SAVED 		0xC0
 	u_int8_t subpage;
 #define	SMS_SUBPAGE_PAGE_0		0x00
 #define	SMS_SUBPAGE_ALL			0xff
 	u_int8_t length;
 	u_int8_t control;
 };
 
 struct scsi_mode_sense_10
 {
 	u_int8_t opcode;
 	u_int8_t byte2;		/* same bits as small version */
 #define	SMS10_LLBAA			0x10
 	u_int8_t page; 		/* same bits as small version */
 	u_int8_t subpage;
 	u_int8_t unused[3];
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 struct scsi_mode_select_6
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SMS_SP	0x01
 #define	SMS_PF	0x10
 	u_int8_t unused[2];
 	u_int8_t length;
 	u_int8_t control;
 };
 
 struct scsi_mode_select_10
 {
 	u_int8_t opcode;
 	u_int8_t byte2;		/* same bits as small version */
 	u_int8_t unused[5];
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 /*
  * When sending a mode select to a tape drive, the medium type must be 0.
  */
 struct scsi_mode_hdr_6
 {
 	u_int8_t datalen;
 	u_int8_t medium_type;
 	u_int8_t dev_specific;
 	u_int8_t block_descr_len;
 };
 
 struct scsi_mode_hdr_10
 {
 	u_int8_t datalen[2];
 	u_int8_t medium_type;
 	u_int8_t dev_specific;
 	u_int8_t reserved[2];
 	u_int8_t block_descr_len[2];
 };
 
 struct scsi_mode_block_descr
 {
 	u_int8_t density_code;
 	u_int8_t num_blocks[3];
 	u_int8_t reserved;
 	u_int8_t block_len[3];
 };
 
 struct scsi_per_res_in
 {
 	u_int8_t opcode;
 	u_int8_t action;
 #define	SPRI_RK	0x00
 #define	SPRI_RR	0x01
 #define	SPRI_RC	0x02
 #define	SPRI_RS	0x03
 	u_int8_t reserved[5];
 	u_int8_t length[2];
 #define	SPRI_MAX_LEN		0xffff
 	u_int8_t control;
 };
 
 struct scsi_per_res_in_header
 {
 	u_int8_t generation[4];
 	u_int8_t length[4];
 };
 
 struct scsi_per_res_key
 {
 	u_int8_t key[8];
 };
 
 struct scsi_per_res_in_keys
 {
 	struct scsi_per_res_in_header header;
 	struct scsi_per_res_key keys[0];
 };
 
 struct scsi_per_res_cap
 {
 	uint8_t length[2];
 	uint8_t flags1;
 #define	SPRI_RLR_C		0x80
 #define	SPRI_CRH		0x10
 #define	SPRI_SIP_C		0x08
 #define	SPRI_ATP_C		0x04
 #define	SPRI_PTPL_C		0x01
 	uint8_t flags2;
 #define	SPRI_TMV		0x80
 #define	SPRI_ALLOW_CMD_MASK	0x70
 #define	SPRI_ALLOW_CMD_SHIFT	4
 #define	SPRI_ALLOW_NA		0x00
 #define	SPRI_ALLOW_1		0x10
 #define	SPRI_ALLOW_2		0x20
 #define	SPRI_ALLOW_3		0x30
 #define	SPRI_ALLOW_4		0x40
 #define	SPRI_ALLOW_5		0x50
 #define	SPRI_PTPL_A		0x01
 	uint8_t type_mask[2];
 #define	SPRI_TM_WR_EX_AR	0x8000
 #define	SPRI_TM_EX_AC_RO	0x4000
 #define	SPRI_TM_WR_EX_RO	0x2000
 #define	SPRI_TM_EX_AC		0x0800
 #define	SPRI_TM_WR_EX		0x0200
 #define	SPRI_TM_EX_AC_AR	0x0001
 	uint8_t reserved[2];
 };
 
 struct scsi_per_res_in_rsrv_data
 {
 	uint8_t reservation[8];
 	uint8_t scope_addr[4];
 	uint8_t reserved;
 	uint8_t scopetype;
 #define	SPRT_WE    0x01
 #define	SPRT_EA    0x03
 #define	SPRT_WERO  0x05
 #define	SPRT_EARO  0x06
 #define	SPRT_WEAR  0x07
 #define	SPRT_EAAR  0x08
 	uint8_t extent_length[2];
 };
 
 struct scsi_per_res_in_rsrv
 {
 	struct scsi_per_res_in_header header;
 	struct scsi_per_res_in_rsrv_data data;
 };
 
 struct scsi_per_res_in_full_desc
 {
 	struct scsi_per_res_key res_key;
 	uint8_t reserved1[4];
 	uint8_t flags;
 #define	SPRI_FULL_ALL_TG_PT	0x02
 #define	SPRI_FULL_R_HOLDER	0x01
 	uint8_t scopetype;
 	uint8_t reserved2[4];
 	uint8_t rel_trgt_port_id[2];
 	uint8_t additional_length[4];
 	uint8_t transport_id[];
 };
 
 struct scsi_per_res_in_full
 {
 	struct scsi_per_res_in_header header;
 	struct scsi_per_res_in_full_desc desc[];
 };
 
 struct scsi_per_res_out
 {
 	u_int8_t opcode;
 	u_int8_t action;
 #define	SPRO_REGISTER		0x00
 #define	SPRO_RESERVE		0x01
 #define	SPRO_RELEASE		0x02
 #define	SPRO_CLEAR		0x03
 #define	SPRO_PREEMPT		0x04
 #define	SPRO_PRE_ABO		0x05
 #define	SPRO_REG_IGNO		0x06
 #define	SPRO_REG_MOVE		0x07
 #define	SPRO_REPL_LOST_RES	0x08
 #define	SPRO_ACTION_MASK	0x1f
 	u_int8_t scope_type;
 #define	SPR_SCOPE_MASK		0xf0
 #define	SPR_SCOPE_SHIFT		4
 #define	SPR_LU_SCOPE		0x00
 #define	SPR_EXTENT_SCOPE	0x10
 #define	SPR_ELEMENT_SCOPE	0x20
 #define	SPR_TYPE_MASK		0x0f
 #define	SPR_TYPE_RD_SHARED	0x00
 #define	SPR_TYPE_WR_EX		0x01
 #define	SPR_TYPE_RD_EX		0x02
 #define	SPR_TYPE_EX_AC		0x03
 #define	SPR_TYPE_SHARED		0x04
 #define	SPR_TYPE_WR_EX_RO	0x05
 #define	SPR_TYPE_EX_AC_RO	0x06
 #define	SPR_TYPE_WR_EX_AR	0x07
 #define	SPR_TYPE_EX_AC_AR	0x08
 	u_int8_t reserved[2];
 	u_int8_t length[4];
 	u_int8_t control;
 };
 
 struct scsi_per_res_out_parms
 {
 	struct scsi_per_res_key res_key;
 	u_int8_t serv_act_res_key[8];
 	u_int8_t scope_spec_address[4];
 	u_int8_t flags;
 #define	SPR_SPEC_I_PT		0x08
 #define	SPR_ALL_TG_PT		0x04
 #define	SPR_APTPL		0x01
 	u_int8_t reserved1;
 	u_int8_t extent_length[2];
 	u_int8_t transport_id_list[];
 };
 
 struct scsi_per_res_out_trans_ids {
 	u_int8_t additional_length[4];
 	u_int8_t transport_ids[];
 };
 
 /*
  * Used with REGISTER AND MOVE serivce action of the PERSISTENT RESERVE OUT
  * command.
  */
 struct scsi_per_res_reg_move
 {
 	struct scsi_per_res_key res_key;
 	u_int8_t serv_act_res_key[8];
 	u_int8_t reserved;
 	u_int8_t flags;
 #define	SPR_REG_MOVE_UNREG	0x02
 #define	SPR_REG_MOVE_APTPL	0x01
 	u_int8_t rel_trgt_port_id[2];
 	u_int8_t transport_id_length[4];
 	u_int8_t transport_id[];
 };
 
 struct scsi_transportid_header
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_FORMAT_MASK		0xc0
 #define	SCSI_TRN_FORMAT_SHIFT		6
 #define	SCSI_TRN_PROTO_MASK		0x0f
 };
 
 struct scsi_transportid_fcp
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_FCP_FORMAT_DEFAULT	0x00
 	uint8_t reserved1[7];
 	uint8_t n_port_name[8];
 	uint8_t reserved2[8];
 };
 
 struct scsi_transportid_spi
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_SPI_FORMAT_DEFAULT	0x00
 	uint8_t reserved1;
 	uint8_t scsi_addr[2];
 	uint8_t obsolete[2];
 	uint8_t rel_trgt_port_id[2];
 	uint8_t reserved2[16];
 };
 
 struct scsi_transportid_1394
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_1394_FORMAT_DEFAULT	0x00
 	uint8_t reserved1[7];
 	uint8_t eui64[8];
 	uint8_t reserved2[8];
 };
 
 struct scsi_transportid_rdma
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_RDMA_FORMAT_DEFAULT	0x00
 	uint8_t reserved[7];
 #define	SCSI_TRN_RDMA_PORT_LEN		16
 	uint8_t initiator_port_id[SCSI_TRN_RDMA_PORT_LEN];
 };
 
 struct scsi_transportid_iscsi_device
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_ISCSI_FORMAT_DEVICE	0x00
 	uint8_t reserved;
 	uint8_t additional_length[2];
 	uint8_t iscsi_name[];
 };
 
 struct scsi_transportid_iscsi_port
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_ISCSI_FORMAT_PORT	0x40
 	uint8_t reserved;
 	uint8_t additional_length[2];
 	uint8_t iscsi_name[];
 	/*
 	 * Followed by a separator and iSCSI initiator session ID
 	 */
 };
 
 struct scsi_transportid_sas
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_SAS_FORMAT_DEFAULT	0x00
 	uint8_t reserved1[3];
 	uint8_t sas_address[8];
 	uint8_t reserved2[12];
 };
 
 struct scsi_sop_routing_id_norm {
 	uint8_t bus;
 	uint8_t devfunc;
 #define	SCSI_TRN_SOP_BUS_MAX		0xff
 #define	SCSI_TRN_SOP_DEV_MAX		0x1f
 #define	SCSI_TRN_SOP_DEV_MASK		0xf8
 #define	SCSI_TRN_SOP_DEV_SHIFT		3
 #define	SCSI_TRN_SOP_FUNC_NORM_MASK	0x07
 #define	SCSI_TRN_SOP_FUNC_NORM_MAX	0x07
 };
 
 struct scsi_sop_routing_id_alt {
 	uint8_t bus;
 	uint8_t function;
 #define	SCSI_TRN_SOP_FUNC_ALT_MAX	0xff
 };
 
 struct scsi_transportid_sop
 {
 	uint8_t format_protocol;
 #define	SCSI_TRN_SOP_FORMAT_DEFAULT	0x00
 	uint8_t reserved1;
 	uint8_t routing_id[2];
 	uint8_t reserved2[20];
 };
 
 struct scsi_log_sense
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SLS_SP				0x01
 #define	SLS_PPC				0x02
 	u_int8_t page;
 #define	SLS_PAGE_CODE 			0x3F
 #define	SLS_SUPPORTED_PAGES_PAGE	0x00
 #define	SLS_OVERRUN_PAGE		0x01
 #define	SLS_ERROR_WRITE_PAGE		0x02
 #define	SLS_ERROR_READ_PAGE		0x03
 #define	SLS_ERROR_READREVERSE_PAGE	0x04
 #define	SLS_ERROR_VERIFY_PAGE		0x05
 #define	SLS_ERROR_NONMEDIUM_PAGE	0x06
 #define	SLS_ERROR_LASTN_PAGE		0x07
 #define	SLS_LOGICAL_BLOCK_PROVISIONING	0x0c
 #define	SLS_SELF_TEST_PAGE		0x10
 #define	SLS_IE_PAGE			0x2f
 #define	SLS_PAGE_CTRL_MASK		0xC0
 #define	SLS_PAGE_CTRL_THRESHOLD		0x00
 #define	SLS_PAGE_CTRL_CUMULATIVE	0x40
 #define	SLS_PAGE_CTRL_THRESH_DEFAULT	0x80
 #define	SLS_PAGE_CTRL_CUMUL_DEFAULT	0xC0
 	u_int8_t subpage;
 #define	SLS_SUPPORTED_SUBPAGES_SUBPAGE	0xff
 	u_int8_t reserved;
 	u_int8_t paramptr[2];
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 struct scsi_log_select
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 /*	SLS_SP				0x01 */
 #define	SLS_PCR				0x02
 	u_int8_t page;
 /*	SLS_PAGE_CTRL_MASK		0xC0 */
 /*	SLS_PAGE_CTRL_THRESHOLD		0x00 */
 /*	SLS_PAGE_CTRL_CUMULATIVE	0x40 */
 /*	SLS_PAGE_CTRL_THRESH_DEFAULT	0x80 */
 /*	SLS_PAGE_CTRL_CUMUL_DEFAULT	0xC0 */
 	u_int8_t reserved[4];
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 struct scsi_log_header
 {
 	u_int8_t page;
 #define	SL_PAGE_CODE			0x3F
 #define	SL_SPF				0x40
 #define	SL_DS				0x80
 	u_int8_t subpage;
 	u_int8_t datalen[2];
 };
 
 struct scsi_log_param_header {
 	u_int8_t param_code[2];
 	u_int8_t param_control;
 #define	SLP_LP				0x01
 #define	SLP_LBIN			0x02
 #define	SLP_TMC_MASK			0x0C
 #define	SLP_TMC_ALWAYS			0x00
 #define	SLP_TMC_EQUAL			0x04
 #define	SLP_TMC_NOTEQUAL		0x08
 #define	SLP_TMC_GREATER			0x0C
 #define	SLP_ETC				0x10
 #define	SLP_TSD				0x20
 #define	SLP_DS				0x40
 #define	SLP_DU				0x80
 	u_int8_t param_len;
 };
 
 struct scsi_control_page {
 	u_int8_t page_code;
 	u_int8_t page_length;
 	u_int8_t rlec;
 #define	SCP_RLEC			0x01	/*Report Log Exception Cond*/
 #define	SCP_GLTSD			0x02	/*Global Logging target
 						  save disable */
 #define	SCP_DSENSE			0x04	/*Descriptor Sense */
 #define	SCP_DPICZ			0x08	/*Disable Prot. Info Check
 						  if Prot. Field is Zero */
 #define	SCP_TMF_ONLY			0x10	/*TM Functions Only*/
 #define	SCP_TST_MASK			0xE0	/*Task Set Type Mask*/
 #define	SCP_TST_ONE			0x00	/*One Task Set*/
 #define	SCP_TST_SEPARATE		0x20	/*Separate Task Sets*/
 	u_int8_t queue_flags;
 #define	SCP_QUEUE_ALG_MASK		0xF0
 #define	SCP_QUEUE_ALG_RESTRICTED	0x00
 #define	SCP_QUEUE_ALG_UNRESTRICTED	0x10
 #define	SCP_NUAR			0x08	/*No UA on release*/
 #define	SCP_QUEUE_ERR			0x02	/*Queued I/O aborted for CACs*/
 #define	SCP_QUEUE_DQUE			0x01	/*Queued I/O disabled*/
 	u_int8_t eca_and_aen;
 #define	SCP_EECA			0x80	/*Enable Extended CA*/
 #define	SCP_RAC				0x40	/*Report a check*/
 #define	SCP_SWP				0x08	/*Software Write Protect*/
 #define	SCP_RAENP			0x04	/*Ready AEN Permission*/
 #define	SCP_UAAENP			0x02	/*UA AEN Permission*/
 #define	SCP_EAENP			0x01	/*Error AEN Permission*/
 	u_int8_t flags4;
 #define	SCP_ATO				0x80	/*Application tag owner*/
 #define	SCP_TAS				0x40	/*Task aborted status*/
 #define	SCP_ATMPE			0x20	/*Application tag mode page*/
 #define	SCP_RWWP			0x10	/*Reject write without prot*/
 	u_int8_t aen_holdoff_period[2];
 	u_int8_t busy_timeout_period[2];
 	u_int8_t extended_selftest_completion_time[2];
 };
 
 struct scsi_cache_page {
 	u_int8_t page_code;
 #define	SCHP_PAGE_SAVABLE		0x80	/* Page is savable */
 	u_int8_t page_length;
 	u_int8_t cache_flags;
 #define	SCHP_FLAGS_WCE			0x04	/* Write Cache Enable */
 #define	SCHP_FLAGS_MF			0x02	/* Multiplication factor */
 #define	SCHP_FLAGS_RCD			0x01	/* Read Cache Disable */
 	u_int8_t rw_cache_policy;
 	u_int8_t dis_prefetch[2];
 	u_int8_t min_prefetch[2];
 	u_int8_t max_prefetch[2];
 	u_int8_t max_prefetch_ceil[2];
 };
 
 /*
  * XXX KDM
  * Updated version of the cache page, as of SBC.  Update this to SBC-3 and
  * rationalize the two.
  */
 struct scsi_caching_page {
 	uint8_t page_code;
 #define	SMS_CACHING_PAGE		0x08
 	uint8_t page_length;
 	uint8_t flags1;
 #define	SCP_IC		0x80
 #define	SCP_ABPF	0x40
 #define	SCP_CAP		0x20
 #define	SCP_DISC	0x10
 #define	SCP_SIZE	0x08
 #define	SCP_WCE		0x04
 #define	SCP_MF		0x02
 #define	SCP_RCD		0x01
 	uint8_t ret_priority;
 	uint8_t disable_pf_transfer_len[2];
 	uint8_t min_prefetch[2];
 	uint8_t max_prefetch[2];
 	uint8_t max_pf_ceiling[2];
 	uint8_t flags2;
 #define	SCP_FSW		0x80
 #define	SCP_LBCSS	0x40
 #define	SCP_DRA		0x20
 #define	SCP_VS1		0x10
 #define	SCP_VS2		0x08
 	uint8_t cache_segments;
 	uint8_t cache_seg_size[2];
 	uint8_t reserved;
 	uint8_t non_cache_seg_size[3];
 };
 
 /*
  * XXX KDM move this off to a vendor shim.
  */
 struct copan_debugconf_subpage {
 	uint8_t page_code;
 #define DBGCNF_PAGE_CODE		0x00
 	uint8_t subpage;
 #define DBGCNF_SUBPAGE_CODE	0xF0
 	uint8_t page_length[2];
 	uint8_t page_version;
 #define DBGCNF_VERSION			0x00
 	uint8_t ctl_time_io_secs[2];
 };
 
 
 struct scsi_info_exceptions_page {
 	u_int8_t page_code;
 #define	SIEP_PAGE_SAVABLE		0x80	/* Page is savable */
 	u_int8_t page_length;
 	u_int8_t info_flags;
 #define	SIEP_FLAGS_PERF			0x80
 #define	SIEP_FLAGS_EBF			0x20
 #define	SIEP_FLAGS_EWASC		0x10
 #define	SIEP_FLAGS_DEXCPT		0x08
 #define	SIEP_FLAGS_TEST			0x04
 #define	SIEP_FLAGS_EBACKERR		0x02
 #define	SIEP_FLAGS_LOGERR		0x01
 	u_int8_t mrie;
 	u_int8_t interval_timer[4];
 	u_int8_t report_count[4];
 };
 
 struct scsi_logical_block_provisioning_page_descr {
 	uint8_t flags;
 #define	SLBPPD_ENABLED		0x80
 #define	SLBPPD_TYPE_MASK	0x38
 #define	SLBPPD_ARMING_MASK	0x07
 #define	SLBPPD_ARMING_DEC	0x02
 #define	SLBPPD_ARMING_INC	0x01
 	uint8_t resource;
 	uint8_t reserved[2];
 	uint8_t count[4];
 };
 
 struct scsi_logical_block_provisioning_page {
 	uint8_t page_code;
 	uint8_t subpage_code;
 	uint8_t page_length[2];
 	uint8_t flags;
 #define	SLBPP_SITUA		0x01
 	uint8_t reserved[11];
 	struct scsi_logical_block_provisioning_page_descr descr[0];
 };
 
 /*
  * SCSI protocol identifier values, current as of SPC4r36l.
  */
 #define	SCSI_PROTO_FC		0x00	/* Fibre Channel */
 #define	SCSI_PROTO_SPI		0x01	/* Parallel SCSI */
 #define	SCSI_PROTO_SSA		0x02	/* Serial Storage Arch. */
 #define	SCSI_PROTO_1394		0x03	/* IEEE 1394 (Firewire) */
 #define	SCSI_PROTO_RDMA		0x04	/* SCSI RDMA Protocol */
 #define	SCSI_PROTO_ISCSI	0x05	/* Internet SCSI */
 #define	SCSI_PROTO_iSCSI	0x05	/* Internet SCSI */
 #define	SCSI_PROTO_SAS		0x06	/* SAS Serial SCSI Protocol */
 #define	SCSI_PROTO_ADT		0x07	/* Automation/Drive Int. Trans. Prot.*/
 #define	SCSI_PROTO_ADITP	0x07	/* Automation/Drive Int. Trans. Prot.*/
 #define	SCSI_PROTO_ATA		0x08	/* AT Attachment Interface */
 #define	SCSI_PROTO_UAS		0x09	/* USB Atached SCSI */
 #define	SCSI_PROTO_SOP		0x0a	/* SCSI over PCI Express */
 #define	SCSI_PROTO_NONE		0x0f	/* No specific protocol */
 
 struct scsi_proto_specific_page {
 	u_int8_t page_code;
 #define	SPSP_PAGE_SAVABLE		0x80	/* Page is savable */
 	u_int8_t page_length;
 	u_int8_t protocol;
 #define	SPSP_PROTO_FC			SCSI_PROTO_FC
 #define	SPSP_PROTO_SPI			SCSI_PROTO_SPI
 #define	SPSP_PROTO_SSA			SCSI_PROTO_SSA
 #define	SPSP_PROTO_1394			SCSI_PROTO_1394
 #define	SPSP_PROTO_RDMA			SCSI_PROTO_RDMA
 #define	SPSP_PROTO_ISCSI		SCSI_PROTO_ISCSI
 #define	SPSP_PROTO_SAS			SCSI_PROTO_SAS
 #define	SPSP_PROTO_ADT			SCSI_PROTO_ADITP
 #define	SPSP_PROTO_ATA			SCSI_PROTO_ATA
 #define	SPSP_PROTO_UAS			SCSI_PROTO_UAS
 #define	SPSP_PROTO_SOP			SCSI_PROTO_SOP
 #define	SPSP_PROTO_NONE			SCSI_PROTO_NONE
 };
 
 struct scsi_reserve
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SR_EXTENT	0x01
 #define	SR_ID_MASK	0x0e
 #define	SR_3RDPTY	0x10
 #define	SR_LUN_MASK	0xe0
 	u_int8_t resv_id;
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 struct scsi_reserve_10 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 #define	SR10_3RDPTY	0x10
 #define	SR10_LONGID	0x02
 #define	SR10_EXTENT	0x01
 	uint8_t resv_id;
 	uint8_t thirdparty_id;
 	uint8_t reserved[3];
 	uint8_t length[2];
 	uint8_t control;
 };
 
 
 struct scsi_release
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 	u_int8_t resv_id;
 	u_int8_t unused[1];
 	u_int8_t length;
 	u_int8_t control;
 };
 
 struct scsi_release_10 {
 	uint8_t opcode;
 	uint8_t byte2;
 	uint8_t resv_id;
 	uint8_t thirdparty_id;
 	uint8_t reserved[3];
 	uint8_t length[2];
 	uint8_t control;
 };
 
 struct scsi_prevent
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 	u_int8_t unused[2];
 	u_int8_t how;
 	u_int8_t control;
 };
 #define	PR_PREVENT 0x01
 #define	PR_ALLOW   0x00
 
 struct scsi_sync_cache
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SSC_IMMED	0x02
 #define	SSC_RELADR	0x01
 	u_int8_t begin_lba[4];
 	u_int8_t reserved;
 	u_int8_t lb_count[2];
 	u_int8_t control;	
 };
 
 struct scsi_sync_cache_16
 {
 	uint8_t opcode;
 	uint8_t byte2;
 	uint8_t begin_lba[8];
 	uint8_t lb_count[4];
 	uint8_t reserved;
 	uint8_t control;
 };
 
 struct scsi_format {
 	uint8_t opcode;
 	uint8_t byte2;
 #define	SF_LONGLIST		0x20
 #define	SF_FMTDATA		0x10
 #define	SF_CMPLIST		0x08
 #define	SF_FORMAT_MASK		0x07
 #define	SF_FORMAT_BLOCK		0x00
 #define	SF_FORMAT_LONG_BLOCK	0x03
 #define	SF_FORMAT_BFI		0x04
 #define	SF_FORMAT_PHYS		0x05
 	uint8_t vendor;
 	uint8_t interleave[2];
 	uint8_t control;
 };
 
 struct scsi_format_header_short {
 	uint8_t reserved;
 #define	SF_DATA_FOV	0x80
 #define	SF_DATA_DPRY	0x40
 #define	SF_DATA_DCRT	0x20
 #define	SF_DATA_STPF	0x10
 #define	SF_DATA_IP	0x08
 #define	SF_DATA_DSP	0x04
 #define	SF_DATA_IMMED	0x02
 #define	SF_DATA_VS	0x01
 	uint8_t byte2;
 	uint8_t defect_list_len[2];
 };
 
 struct scsi_format_header_long {
 	uint8_t reserved;
 	uint8_t byte2;
 	uint8_t reserved2[2];
 	uint8_t defect_list_len[4];
 };
 
 struct scsi_changedef
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 	u_int8_t unused1;
 	u_int8_t how;
 	u_int8_t unused[4];
 	u_int8_t datalen;
 	u_int8_t control;
 };
 
 struct scsi_read_buffer
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	RWB_MODE		0x1F
 #define	RWB_MODE_HDR_DATA	0x00
 #define	RWB_MODE_VENDOR		0x01
 #define	RWB_MODE_DATA		0x02
 #define	RWB_MODE_DESCR		0x03
 #define	RWB_MODE_DOWNLOAD	0x04
 #define	RWB_MODE_DOWNLOAD_SAVE	0x05
 #define	RWB_MODE_ECHO		0x0A
 #define	RWB_MODE_ECHO_DESCR	0x0B
 #define	RWB_MODE_ERROR_HISTORY	0x1C
         u_int8_t buffer_id;
         u_int8_t offset[3];
         u_int8_t length[3];
         u_int8_t control;
 };
 
 struct scsi_write_buffer
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 	u_int8_t buffer_id;
 	u_int8_t offset[3];
 	u_int8_t length[3];
 	u_int8_t control;
 };
 
 struct scsi_rw_6
 {
 	u_int8_t opcode;
 	u_int8_t addr[3];
 /* only 5 bits are valid in the MSB address byte */
 #define	SRW_TOPADDR	0x1F
 	u_int8_t length;
 	u_int8_t control;
 };
 
 struct scsi_rw_10
 {
 	u_int8_t opcode;
 #define	SRW10_RELADDR	0x01
 /* EBP defined for WRITE(10) only */
 #define	SRW10_EBP	0x04
 #define	SRW10_FUA	0x08
 #define	SRW10_DPO	0x10
 	u_int8_t byte2;
 	u_int8_t addr[4];
 	u_int8_t reserved;
 	u_int8_t length[2];
 	u_int8_t control;
 };
 
 struct scsi_rw_12
 {
 	u_int8_t opcode;
 #define	SRW12_RELADDR	0x01
 #define	SRW12_FUA	0x08
 #define	SRW12_DPO	0x10
 	u_int8_t byte2;
 	u_int8_t addr[4];
 	u_int8_t length[4];
 	u_int8_t reserved;
 	u_int8_t control;
 };
 
 struct scsi_rw_16
 {
 	u_int8_t opcode;
 #define	SRW16_RELADDR	0x01
 #define	SRW16_FUA	0x08
 #define	SRW16_DPO	0x10
 	u_int8_t byte2;
 	u_int8_t addr[8];
 	u_int8_t length[4];
 	u_int8_t reserved;
 	u_int8_t control;
 };
 
 struct scsi_write_same_10
 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 #define	SWS_LBDATA	0x02
 #define	SWS_PBDATA	0x04
 #define	SWS_UNMAP	0x08
 #define	SWS_ANCHOR	0x10
 	uint8_t	addr[4];
 	uint8_t	group;
 	uint8_t	length[2];
 	uint8_t	control;
 };
 
 struct scsi_write_same_16
 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 #define	SWS_NDOB	0x01
 	uint8_t	addr[8];
 	uint8_t	length[4];
 	uint8_t	group;
 	uint8_t	control;
 };
 
 struct scsi_unmap
 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 #define	SU_ANCHOR	0x01
 	uint8_t	reserved[4];
 	uint8_t	group;
 	uint8_t	length[2];
 	uint8_t	control;
 };
 
 struct scsi_unmap_header
 {
 	uint8_t	length[2];
 	uint8_t	desc_length[2];
 	uint8_t	reserved[4];
 };
 
 struct scsi_unmap_desc
 {
 	uint8_t	lba[8];
 	uint8_t	length[4];
 	uint8_t	reserved[4];
 };
 
 struct scsi_write_verify_10
 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 #define	SWV_BYTCHK		0x02
 #define	SWV_DPO			0x10
 #define	SWV_WRPROECT_MASK	0xe0
 	uint8_t	addr[4];
 	uint8_t	group;
 	uint8_t length[2];
 	uint8_t	control;
 };
 
 struct scsi_write_verify_12
 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 	uint8_t	addr[4];
 	uint8_t	length[4];
 	uint8_t	group;
 	uint8_t	control;
 };
 
 struct scsi_write_verify_16
 {
 	uint8_t	opcode;
 	uint8_t	byte2;
 	uint8_t	addr[8];
 	uint8_t	length[4];
 	uint8_t	group;
 	uint8_t	control;
 };
 
 
 struct scsi_start_stop_unit
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SSS_IMMED		0x01
 	u_int8_t reserved[2];
 	u_int8_t how;
 #define	SSS_START		0x01
 #define	SSS_LOEJ		0x02
 #define	SSS_PC_MASK		0xf0
 #define	SSS_PC_START_VALID	0x00
 #define	SSS_PC_ACTIVE		0x10
 #define	SSS_PC_IDLE		0x20
 #define	SSS_PC_STANDBY		0x30
 #define	SSS_PC_LU_CONTROL	0x70
 #define	SSS_PC_FORCE_IDLE_0	0xa0
 #define	SSS_PC_FORCE_STANDBY_0	0xb0
 	u_int8_t control;
 };
 
 struct ata_pass_12 {
 	u_int8_t opcode;
 	u_int8_t protocol;
 #define	AP_PROTO_HARD_RESET	(0x00 << 1)
 #define	AP_PROTO_SRST		(0x01 << 1)
 #define	AP_PROTO_NON_DATA	(0x03 << 1)
 #define	AP_PROTO_PIO_IN		(0x04 << 1)
 #define	AP_PROTO_PIO_OUT	(0x05 << 1)
 #define	AP_PROTO_DMA		(0x06 << 1)
 #define	AP_PROTO_DMA_QUEUED	(0x07 << 1)
 #define	AP_PROTO_DEVICE_DIAG	(0x08 << 1)
 #define	AP_PROTO_DEVICE_RESET	(0x09 << 1)
 #define	AP_PROTO_UDMA_IN	(0x0a << 1)
 #define	AP_PROTO_UDMA_OUT	(0x0b << 1)
 #define	AP_PROTO_FPDMA		(0x0c << 1)
 #define	AP_PROTO_RESP_INFO	(0x0f << 1)
 #define	AP_MULTI	0xe0
 	u_int8_t flags;
 #define	AP_T_LEN	0x03
 #define	AP_BB		0x04
 #define	AP_T_DIR	0x08
 #define	AP_CK_COND	0x20
 #define	AP_OFFLINE	0x60
 	u_int8_t features;
 	u_int8_t sector_count;
 	u_int8_t lba_low;
 	u_int8_t lba_mid;
 	u_int8_t lba_high;
 	u_int8_t device;
 	u_int8_t command;
 	u_int8_t reserved;
 	u_int8_t control;
 };
 
 struct scsi_maintenance_in
 {
         uint8_t  opcode;
         uint8_t  byte2;
 #define SERVICE_ACTION_MASK  0x1f
 #define SA_RPRT_TRGT_GRP     0x0a
         uint8_t  reserved[4];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_report_supported_opcodes
 {
         uint8_t  opcode;
         uint8_t  service_action;
         uint8_t  options;
 #define RSO_RCTD		0x80
 #define RSO_OPTIONS_MASK	0x07
 #define RSO_OPTIONS_ALL		0x00
 #define RSO_OPTIONS_OC		0x01
 #define RSO_OPTIONS_OC_SA	0x02
         uint8_t  requested_opcode;
         uint8_t  requested_service_action[2];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_report_supported_opcodes_timeout
 {
 	uint8_t  length[2];
 	uint8_t  reserved;
 	uint8_t  cmd_specific;
 	uint8_t  nominal_time[4];
 	uint8_t  recommended_time[4];
 };
 
 struct scsi_report_supported_opcodes_descr
 {
 	uint8_t  opcode;
 	uint8_t  reserved;
 	uint8_t  service_action[2];
 	uint8_t  reserved2;
 	uint8_t  flags;
 #define RSO_SERVACTV		0x01
 #define RSO_CTDP		0x02
 	uint8_t  cdb_length[2];
 	struct scsi_report_supported_opcodes_timeout timeout[0];
 };
 
 struct scsi_report_supported_opcodes_all
 {
 	uint8_t  length[4];
 	struct scsi_report_supported_opcodes_descr descr[0];
 };
 
 struct scsi_report_supported_opcodes_one
 {
 	uint8_t  reserved;
 	uint8_t  support;
 #define RSO_ONE_CTDP		0x80
 	uint8_t  cdb_length[2];
 	uint8_t  cdb_usage[];
 };
 
 struct scsi_report_supported_tmf
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 	uint8_t  reserved[4];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_report_supported_tmf_data
 {
 	uint8_t  byte1;
 #define RST_WAKES		0x01
 #define RST_TRS			0x02
 #define RST_QTS			0x04
 #define RST_LURS		0x08
 #define RST_CTSS		0x10
 #define RST_CACAS		0x20
 #define RST_ATSS		0x40
 #define RST_ATS			0x80
 	uint8_t  byte2;
 #define RST_ITNRS		0x01
 #define RST_QTSS		0x02
 #define RST_QAES		0x04
 	uint8_t  reserved[2];
 };
 
 struct scsi_report_timestamp
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 	uint8_t  reserved[4];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_report_timestamp_data
 {
 	uint8_t  length[2];
 	uint8_t  origin;
 #define RTS_ORIG_MASK		0x00
 #define RTS_ORIG_ZERO		0x00
 #define RTS_ORIG_SET		0x02
 #define RTS_ORIG_OUTSIDE	0x03
 	uint8_t  reserved;
 	uint8_t  timestamp[6];
 	uint8_t  reserve2[2];
 };
 
 struct scsi_receive_copy_status_lid1
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define RCS_RCS_LID1		0x00
 	uint8_t  list_identifier;
 	uint8_t  reserved[7];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_receive_copy_status_lid1_data
 {
 	uint8_t  available_data[4];
 	uint8_t  copy_command_status;
 #define RCS_CCS_INPROG		0x00
 #define RCS_CCS_COMPLETED	0x01
 #define RCS_CCS_ERROR		0x02
 	uint8_t  segments_processed[2];
 	uint8_t  transfer_count_units;
 #define RCS_TC_BYTES		0x00
 #define RCS_TC_KBYTES		0x01
 #define RCS_TC_MBYTES		0x02
 #define RCS_TC_GBYTES		0x03
 #define RCS_TC_TBYTES		0x04
 #define RCS_TC_PBYTES		0x05
 #define RCS_TC_EBYTES		0x06
 #define RCS_TC_LBAS		0xf1
 	uint8_t  transfer_count[4];
 };
 
 struct scsi_receive_copy_failure_details
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define RCS_RCFD		0x04
 	uint8_t  list_identifier;
 	uint8_t  reserved[7];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_receive_copy_failure_details_data
 {
 	uint8_t  available_data[4];
 	uint8_t  reserved[52];
 	uint8_t  copy_command_status;
 	uint8_t  reserved2;
 	uint8_t  sense_data_length[2];
 	uint8_t  sense_data[];
 };
 
 struct scsi_receive_copy_status_lid4
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define RCS_RCS_LID4		0x05
 	uint8_t  list_identifier[4];
 	uint8_t  reserved[4];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_receive_copy_status_lid4_data
 {
 	uint8_t  available_data[4];
 	uint8_t  response_to_service_action;
 	uint8_t  copy_command_status;
 #define RCS_CCS_COMPLETED_PROD	0x03
 #define RCS_CCS_COMPLETED_RESID	0x04
 #define RCS_CCS_INPROG_FGBG	0x10
 #define RCS_CCS_INPROG_FG	0x11
 #define RCS_CCS_INPROG_BG	0x12
 #define RCS_CCS_ABORTED		0x60
 	uint8_t  operation_counter[2];
 	uint8_t  estimated_status_update_delay[4];
 	uint8_t  extended_copy_completion_status;
 	uint8_t  length_of_the_sense_data_field;
 	uint8_t  sense_data_length;
 	uint8_t  transfer_count_units;
 	uint8_t  transfer_count[8];
 	uint8_t  segments_processed[2];
 	uint8_t  reserved[6];
 	uint8_t  sense_data[];
 };
 
 struct scsi_receive_copy_operating_parameters
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define RCS_RCOP		0x03
 	uint8_t  reserved[8];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_receive_copy_operating_parameters_data
 {
 	uint8_t  length[4];
 	uint8_t  snlid;
 #define RCOP_SNLID		0x01
 	uint8_t  reserved[3];
 	uint8_t  maximum_cscd_descriptor_count[2];
 	uint8_t  maximum_segment_descriptor_count[2];
 	uint8_t  maximum_descriptor_list_length[4];
 	uint8_t  maximum_segment_length[4];
 	uint8_t  maximum_inline_data_length[4];
 	uint8_t  held_data_limit[4];
 	uint8_t  maximum_stream_device_transfer_size[4];
 	uint8_t  reserved2[2];
 	uint8_t  total_concurrent_copies[2];
 	uint8_t  maximum_concurrent_copies;
 	uint8_t  data_segment_granularity;
 	uint8_t  inline_data_granularity;
 	uint8_t  held_data_granularity;
 	uint8_t  reserved3[3];
 	uint8_t  implemented_descriptor_list_length;
 	uint8_t  list_of_implemented_descriptor_type_codes[0];
 };
 
 struct scsi_extended_copy
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define EC_EC_LID1		0x00
 #define EC_EC_LID4		0x01
 	uint8_t  reserved[8];
 	uint8_t  length[4];
 	uint8_t  reserved1;
 	uint8_t  control;
 };
 
 struct scsi_ec_cscd_dtsp
 {
 	uint8_t  flags;
 #define EC_CSCD_FIXED		0x01
 #define EC_CSCD_PAD		0x04
 	uint8_t  block_length[3];
 };
 
 struct scsi_ec_cscd
 {
 	uint8_t  type_code;
 #define EC_CSCD_EXT		0xff
 	uint8_t  luidt_pdt;
 #define EC_LUIDT_MASK		0xc0
 #define EC_LUIDT_LUN		0x00
 #define EC_LUIDT_PROXY_TOKEN	0x40
 	uint8_t  relative_initiator_port[2];
 	uint8_t  cscd_params[24];
 	struct scsi_ec_cscd_dtsp dtsp;
 };
 
 struct scsi_ec_cscd_id
 {
 	uint8_t  type_code;
 #define EC_CSCD_ID		0xe4
 	uint8_t  luidt_pdt;
 	uint8_t  relative_initiator_port[2];
 	uint8_t  codeset;
 	uint8_t  id_type;
 	uint8_t  reserved;
 	uint8_t  length;
 	uint8_t  designator[20];
 	struct scsi_ec_cscd_dtsp dtsp;
 };
 
 struct scsi_ec_segment
 {
 	uint8_t  type_code;
 	uint8_t  flags;
 #define EC_SEG_DC		0x02
 #define EC_SEG_CAT		0x01
 	uint8_t  descr_length[2];
 	uint8_t  params[];
 };
 
 struct scsi_ec_segment_b2b
 {
 	uint8_t  type_code;
 #define EC_SEG_B2B		0x02
 	uint8_t  flags;
 	uint8_t  descr_length[2];
 	uint8_t  src_cscd[2];
 	uint8_t  dst_cscd[2];
 	uint8_t  reserved[2];
 	uint8_t  number_of_blocks[2];
 	uint8_t  src_lba[8];
 	uint8_t  dst_lba[8];
 };
 
 struct scsi_ec_segment_verify
 {
 	uint8_t  type_code;
 #define EC_SEG_VERIFY		0x07
 	uint8_t  reserved;
 	uint8_t  descr_length[2];
 	uint8_t  src_cscd[2];
 	uint8_t  reserved2[2];
 	uint8_t  tur;
 	uint8_t  reserved3[3];
 };
 
 struct scsi_ec_segment_register_key
 {
 	uint8_t  type_code;
 #define EC_SEG_REGISTER_KEY	0x14
 	uint8_t  reserved;
 	uint8_t  descr_length[2];
 	uint8_t  reserved2[2];
 	uint8_t  dst_cscd[2];
 	uint8_t  res_key[8];
 	uint8_t  sa_res_key[8];
 	uint8_t  reserved3[4];
 };
 
 struct scsi_extended_copy_lid1_data
 {
 	uint8_t  list_identifier;
 	uint8_t  flags;
 #define EC_PRIORITY		0x07
 #define EC_LIST_ID_USAGE_MASK	0x18
 #define EC_LIST_ID_USAGE_FULL	0x08
 #define EC_LIST_ID_USAGE_NOHOLD	0x10
 #define EC_LIST_ID_USAGE_NONE	0x18
 #define EC_STR			0x20
 	uint8_t  cscd_list_length[2];
 	uint8_t  reserved[4];
 	uint8_t  segment_list_length[4];
 	uint8_t  inline_data_length[4];
 	uint8_t  data[];
 };
 
 struct scsi_extended_copy_lid4_data
 {
 	uint8_t  list_format;
 #define EC_LIST_FORMAT		0x01
 	uint8_t  flags;
 	uint8_t  header_cscd_list_length[2];
 	uint8_t  reserved[11];
 	uint8_t  flags2;
 #define EC_IMMED		0x01
 #define EC_G_SENSE		0x02
 	uint8_t  header_cscd_type_code;
 	uint8_t  reserved2[3];
 	uint8_t  list_identifier[4];
 	uint8_t  reserved3[18];
 	uint8_t  cscd_list_length[2];
 	uint8_t  segment_list_length[2];
 	uint8_t  inline_data_length[2];
 	uint8_t  data[];
 };
 
 struct scsi_copy_operation_abort
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define EC_COA			0x1c
 	uint8_t  list_identifier[4];
 	uint8_t  reserved[9];
 	uint8_t  control;
 };
 
 struct scsi_populate_token
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define EC_PT			0x10
 	uint8_t  reserved[4];
 	uint8_t  list_identifier[4];
 	uint8_t  length[4];
 	uint8_t  group_number;
 	uint8_t  control;
 };
 
 struct scsi_range_desc
 {
 	uint8_t	lba[8];
 	uint8_t	length[4];
 	uint8_t	reserved[4];
 };
 
 struct scsi_populate_token_data
 {
 	uint8_t  length[2];
 	uint8_t  flags;
 #define EC_PT_IMMED			0x01
 #define EC_PT_RTV			0x02
 	uint8_t  reserved;
 	uint8_t  inactivity_timeout[4];
 	uint8_t  rod_type[4];
 	uint8_t  reserved2[2];
 	uint8_t  range_descriptor_length[2];
 	struct scsi_range_desc desc[];
 };
 
 struct scsi_write_using_token
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define EC_WUT			0x11
 	uint8_t  reserved[4];
 	uint8_t  list_identifier[4];
 	uint8_t  length[4];
 	uint8_t  group_number;
 	uint8_t  control;
 };
 
 struct scsi_write_using_token_data
 {
 	uint8_t  length[2];
 	uint8_t  flags;
 #define EC_WUT_IMMED			0x01
 #define EC_WUT_DEL_TKN			0x02
 	uint8_t  reserved[5];
 	uint8_t  offset_into_rod[8];
 	uint8_t  rod_token[512];
 	uint8_t  reserved2[6];
 	uint8_t  range_descriptor_length[2];
 	struct scsi_range_desc desc[];
 };
 
 struct scsi_receive_rod_token_information
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define RCS_RRTI		0x07
 	uint8_t  list_identifier[4];
 	uint8_t  reserved[4];
 	uint8_t  length[4];
 	uint8_t  reserved2;
 	uint8_t  control;
 };
 
 struct scsi_token
 {
 	uint8_t  type[4];
 #define ROD_TYPE_INTERNAL	0x00000000
 #define ROD_TYPE_AUR		0x00010000
 #define ROD_TYPE_PIT_DEF	0x00800000
 #define ROD_TYPE_PIT_VULN	0x00800001
 #define ROD_TYPE_PIT_PERS	0x00800002
 #define ROD_TYPE_PIT_ANY	0x0080FFFF
 #define ROD_TYPE_BLOCK_ZERO	0xFFFF0001
 	uint8_t  reserved[2];
 	uint8_t  length[2];
 	uint8_t  body[0];
 };
 
 struct scsi_report_all_rod_tokens
 {
 	uint8_t  opcode;
 	uint8_t  service_action;
 #define RCS_RART		0x08
 	uint8_t  reserved[8];
 	uint8_t  length[4];
 	uint8_t  reserved2;
 	uint8_t  control;
 };
 
 struct scsi_report_all_rod_tokens_data
 {
 	uint8_t  available_data[4];
 	uint8_t  reserved[4];
 	uint8_t  rod_management_token_list[];
 };
 
 struct ata_pass_16 {
 	u_int8_t opcode;
 	u_int8_t protocol;
 #define	AP_EXTEND	0x01
 	u_int8_t flags;
 #define	AP_FLAG_TLEN_NO_DATA	(0 << 0)
 #define	AP_FLAG_TLEN_FEAT	(1 << 0)
 #define	AP_FLAG_TLEN_SECT_CNT	(2 << 0)
 #define	AP_FLAG_TLEN_STPSIU	(3 << 0)
 #define	AP_FLAG_BYT_BLOK_BYTES	(0 << 2)  
 #define	AP_FLAG_BYT_BLOK_BLOCKS	(1 << 2)  
 #define	AP_FLAG_TDIR_TO_DEV	(0 << 3)  
 #define	AP_FLAG_TDIR_FROM_DEV	(1 << 3)  
 #define	AP_FLAG_CHK_COND	(1 << 5)  
 	u_int8_t features_ext;
 	u_int8_t features;
 	u_int8_t sector_count_ext;
 	u_int8_t sector_count;
 	u_int8_t lba_low_ext;
 	u_int8_t lba_low;
 	u_int8_t lba_mid_ext;
 	u_int8_t lba_mid;
 	u_int8_t lba_high_ext;
 	u_int8_t lba_high;
 	u_int8_t device;
 	u_int8_t command;
 	u_int8_t control;
 };
 
 #define	SC_SCSI_1 0x01
 #define	SC_SCSI_2 0x03
 
 /*
  * Opcodes
  */
 
 #define	TEST_UNIT_READY		0x00
 #define	REQUEST_SENSE		0x03
 #define	READ_6			0x08
 #define	WRITE_6			0x0A
 #define	INQUIRY			0x12
 #define	MODE_SELECT_6		0x15
 #define	MODE_SENSE_6		0x1A
 #define	START_STOP_UNIT		0x1B
 #define	START_STOP		0x1B
 #define	RESERVE      		0x16
 #define	RELEASE      		0x17
 #define	RECEIVE_DIAGNOSTIC	0x1C
 #define	SEND_DIAGNOSTIC		0x1D
 #define	PREVENT_ALLOW		0x1E
 #define	READ_CAPACITY		0x25
 #define	READ_10			0x28
 #define	WRITE_10		0x2A
 #define	POSITION_TO_ELEMENT	0x2B
 #define	WRITE_VERIFY_10		0x2E
 #define	VERIFY_10		0x2F
 #define	SYNCHRONIZE_CACHE	0x35
 #define	READ_DEFECT_DATA_10	0x37
 #define	WRITE_BUFFER            0x3B
 #define	READ_BUFFER             0x3C
 #define	CHANGE_DEFINITION	0x40
 #define	WRITE_SAME_10		0x41
 #define	UNMAP			0x42
 #define	LOG_SELECT		0x4C
 #define	LOG_SENSE		0x4D
 #define	MODE_SELECT_10		0x55
 #define	RESERVE_10		0x56
 #define	RELEASE_10		0x57
 #define	MODE_SENSE_10		0x5A
 #define	PERSISTENT_RES_IN	0x5E
 #define	PERSISTENT_RES_OUT	0x5F
 #define	EXTENDED_COPY		0x83
 #define	RECEIVE_COPY_STATUS	0x84
 #define	ATA_PASS_16		0x85
 #define	READ_16			0x88
 #define	COMPARE_AND_WRITE	0x89
 #define	WRITE_16		0x8A
 #define	WRITE_VERIFY_16		0x8E
 #define	VERIFY_16		0x8F
 #define	SYNCHRONIZE_CACHE_16	0x91
 #define	WRITE_SAME_16		0x93
 #define	WRITE_ATOMIC_16		0x9C
 #define	SERVICE_ACTION_IN	0x9E
 #define	REPORT_LUNS		0xA0
 #define	ATA_PASS_12		0xA1
 #define	MAINTENANCE_IN		0xA3
 #define	MAINTENANCE_OUT		0xA4
 #define	MOVE_MEDIUM     	0xA5
 #define	READ_12			0xA8
 #define	WRITE_12		0xAA
 #define	WRITE_VERIFY_12		0xAE
 #define	VERIFY_12		0xAF
 #define	READ_ELEMENT_STATUS	0xB8
 #define	READ_CD			0xBE
 
 /* Maintenance In Service Action Codes */
 #define	REPORT_IDENTIFYING_INFRMATION		0x05
 #define	REPORT_TARGET_PORT_GROUPS		0x0A
 #define	REPORT_ALIASES				0x0B
 #define	REPORT_SUPPORTED_OPERATION_CODES	0x0C
 #define	REPORT_SUPPORTED_TASK_MANAGEMENT_FUNCTIONS	0x0D
 #define	REPORT_PRIORITY				0x0E
 #define	REPORT_TIMESTAMP			0x0F
 #define	MANAGEMENT_PROTOCOL_IN			0x10
 /* Maintenance Out Service Action Codes */
 #define	SET_IDENTIFY_INFORMATION		0x06
 #define	SET_TARGET_PORT_GROUPS			0x0A
 #define	CHANGE_ALIASES				0x0B
 #define	SET_PRIORITY				0x0E
 #define	SET_TIMESTAMP				0x0F
 #define	MANGAEMENT_PROTOCOL_OUT			0x10
 
 /*
  * Device Types
  */
 #define	T_DIRECT	0x00
 #define	T_SEQUENTIAL	0x01
 #define	T_PRINTER	0x02
 #define	T_PROCESSOR	0x03
 #define	T_WORM		0x04
 #define	T_CDROM		0x05
 #define	T_SCANNER	0x06
 #define	T_OPTICAL 	0x07
 #define	T_CHANGER	0x08
 #define	T_COMM		0x09
 #define	T_ASC0		0x0a
 #define	T_ASC1		0x0b
 #define	T_STORARRAY	0x0c
 #define	T_ENCLOSURE	0x0d
 #define	T_RBC		0x0e
 #define	T_OCRW		0x0f
 #define	T_OSD		0x11
 #define	T_ADC		0x12
 #define	T_NODEVICE	0x1f
 #define	T_ANY		0xff	/* Used in Quirk table matches */
 
 #define	T_REMOV		1
 #define	T_FIXED		0
 
 /*
  * This length is the initial inquiry length used by the probe code, as    
  * well as the length necessary for scsi_print_inquiry() to function 
  * correctly.  If either use requires a different length in the future, 
  * the two values should be de-coupled.
  */
 #define	SHORT_INQUIRY_LENGTH	36
 
 struct scsi_inquiry_data
 {
 	u_int8_t device;
 #define	SID_TYPE(inq_data) ((inq_data)->device & 0x1f)
 #define	SID_QUAL(inq_data) (((inq_data)->device & 0xE0) >> 5)
 #define	SID_QUAL_LU_CONNECTED	0x00	/*
 					 * The specified peripheral device
 					 * type is currently connected to
 					 * logical unit.  If the target cannot
 					 * determine whether or not a physical
 					 * device is currently connected, it
 					 * shall also use this peripheral
 					 * qualifier when returning the INQUIRY
 					 * data.  This peripheral qualifier
 					 * does not mean that the device is
 					 * ready for access by the initiator.
 					 */
 #define	SID_QUAL_LU_OFFLINE	0x01	/*
 					 * The target is capable of supporting
 					 * the specified peripheral device type
 					 * on this logical unit; however, the
 					 * physical device is not currently
 					 * connected to this logical unit.
 					 */
 #define	SID_QUAL_RSVD		0x02
 #define	SID_QUAL_BAD_LU		0x03	/*
 					 * The target is not capable of
 					 * supporting a physical device on
 					 * this logical unit. For this
 					 * peripheral qualifier the peripheral
 					 * device type shall be set to 1Fh to
 					 * provide compatibility with previous
 					 * versions of SCSI. All other
 					 * peripheral device type values are
 					 * reserved for this peripheral
 					 * qualifier.
 					 */
 #define	SID_QUAL_IS_VENDOR_UNIQUE(inq_data) ((SID_QUAL(inq_data) & 0x04) != 0)
 	u_int8_t dev_qual2;
 #define	SID_QUAL2	0x7F
 #define	SID_LU_CONG	0x40
 #define	SID_RMB		0x80
 #define	SID_IS_REMOVABLE(inq_data) (((inq_data)->dev_qual2 & SID_RMB) != 0)
 	u_int8_t version;
 #define	SID_ANSI_REV(inq_data) ((inq_data)->version & 0x07)
 #define		SCSI_REV_0		0
 #define		SCSI_REV_CCS		1
 #define		SCSI_REV_2		2
 #define		SCSI_REV_SPC		3
 #define		SCSI_REV_SPC2		4
 #define		SCSI_REV_SPC3		5
 #define		SCSI_REV_SPC4		6
 
 #define	SID_ECMA	0x38
 #define	SID_ISO		0xC0
 	u_int8_t response_format;
 #define	SID_AENC	0x80
 #define	SID_TrmIOP	0x40
 #define	SID_NormACA	0x20
 #define	SID_HiSup	0x10
 	u_int8_t additional_length;
 #define	SID_ADDITIONAL_LENGTH(iqd)					\
 	((iqd)->additional_length +					\
 	__offsetof(struct scsi_inquiry_data, additional_length) + 1)
 	u_int8_t spc3_flags;
 #define	SPC3_SID_PROTECT	0x01
 #define	SPC3_SID_3PC		0x08
 #define	SPC3_SID_TPGS_MASK	0x30
 #define	SPC3_SID_TPGS_IMPLICIT	0x10
 #define	SPC3_SID_TPGS_EXPLICIT	0x20
 #define	SPC3_SID_ACC		0x40
 #define	SPC3_SID_SCCS		0x80
 	u_int8_t spc2_flags;
 #define	SPC2_SID_ADDR16		0x01
 #define	SPC2_SID_MChngr 	0x08
 #define	SPC2_SID_MultiP 	0x10
 #define	SPC2_SID_EncServ	0x40
 #define	SPC2_SID_BQueue		0x80
 
 #define	INQ_DATA_TQ_ENABLED(iqd)				\
     ((SID_ANSI_REV(iqd) < SCSI_REV_SPC2)? ((iqd)->flags & SID_CmdQue) :	\
     (((iqd)->flags & SID_CmdQue) && !((iqd)->spc2_flags & SPC2_SID_BQueue)) || \
     (!((iqd)->flags & SID_CmdQue) && ((iqd)->spc2_flags & SPC2_SID_BQueue)))
 
 	u_int8_t flags;
 #define	SID_SftRe	0x01
 #define	SID_CmdQue	0x02
 #define	SID_Linked	0x08
 #define	SID_Sync	0x10
 #define	SID_WBus16	0x20
 #define	SID_WBus32	0x40
 #define	SID_RelAdr	0x80
 #define	SID_VENDOR_SIZE   8
 	char	 vendor[SID_VENDOR_SIZE];
 #define	SID_PRODUCT_SIZE  16
 	char	 product[SID_PRODUCT_SIZE];
 #define	SID_REVISION_SIZE 4
 	char	 revision[SID_REVISION_SIZE];
 	/*
 	 * The following fields were taken from SCSI Primary Commands - 2
 	 * (SPC-2) Revision 14, Dated 11 November 1999
 	 */
 #define	SID_VENDOR_SPECIFIC_0_SIZE	20
 	u_int8_t vendor_specific0[SID_VENDOR_SPECIFIC_0_SIZE];
 	/*
 	 * An extension of SCSI Parallel Specific Values
 	 */
 #define	SID_SPI_IUS		0x01
 #define	SID_SPI_QAS		0x02
 #define	SID_SPI_CLOCK_ST	0x00
 #define	SID_SPI_CLOCK_DT	0x04
 #define	SID_SPI_CLOCK_DT_ST	0x0C
 #define	SID_SPI_MASK		0x0F
 	u_int8_t spi3data;
 	u_int8_t reserved2;
 	/*
 	 * Version Descriptors, stored 2 byte values.
 	 */
 	u_int8_t version1[2];
 	u_int8_t version2[2];
 	u_int8_t version3[2];
 	u_int8_t version4[2];
 	u_int8_t version5[2];
 	u_int8_t version6[2];
 	u_int8_t version7[2];
 	u_int8_t version8[2];
 
 	u_int8_t reserved3[22];
 
 #define	SID_VENDOR_SPECIFIC_1_SIZE	160
 	u_int8_t vendor_specific1[SID_VENDOR_SPECIFIC_1_SIZE];
 };
 
 /*
  * This structure is more suited to initiator operation, because the
  * maximum number of supported pages is already allocated.
  */
 struct scsi_vpd_supported_page_list
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define	SVPD_SUPPORTED_PAGE_LIST	0x00
 #define	SVPD_SUPPORTED_PAGES_HDR_LEN	4
 	u_int8_t reserved;
 	u_int8_t length;	/* number of VPD entries */
 #define	SVPD_SUPPORTED_PAGES_SIZE	251
 	u_int8_t list[SVPD_SUPPORTED_PAGES_SIZE];
 };
 
 /*
  * This structure is more suited to target operation, because the
  * number of supported pages is left to the user to allocate.
  */
 struct scsi_vpd_supported_pages
 {
 	u_int8_t device;
 	u_int8_t page_code;
 	u_int8_t reserved;
 #define	SVPD_SUPPORTED_PAGES	0x00
 	u_int8_t length;
 	u_int8_t page_list[0];
 };
 
 
 struct scsi_vpd_unit_serial_number
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define	SVPD_UNIT_SERIAL_NUMBER	0x80
 	u_int8_t reserved;
 	u_int8_t length; /* serial number length */
 #define	SVPD_SERIAL_NUM_SIZE 251
 	u_int8_t serial_num[SVPD_SERIAL_NUM_SIZE];
 };
 
 struct scsi_vpd_device_id
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define	SVPD_DEVICE_ID			0x83
 #define	SVPD_DEVICE_ID_MAX_SIZE		252
 #define	SVPD_DEVICE_ID_HDR_LEN \
     __offsetof(struct scsi_vpd_device_id, desc_list)
 	u_int8_t length[2];
 	u_int8_t desc_list[];
 };
 
 struct scsi_vpd_id_descriptor
 {
 	u_int8_t	proto_codeset;
 	/*
 	 * See the SCSI_PROTO definitions above for the protocols.
 	 */
 #define	SVPD_ID_PROTO_SHIFT	4
 #define	SVPD_ID_CODESET_BINARY	0x01
 #define	SVPD_ID_CODESET_ASCII	0x02
 #define	SVPD_ID_CODESET_UTF8	0x03
 #define	SVPD_ID_CODESET_MASK	0x0f
 	u_int8_t	id_type;
 #define	SVPD_ID_PIV		0x80
 #define	SVPD_ID_ASSOC_LUN	0x00
 #define	SVPD_ID_ASSOC_PORT	0x10
 #define	SVPD_ID_ASSOC_TARGET	0x20
 #define	SVPD_ID_ASSOC_MASK	0x30
 #define	SVPD_ID_TYPE_VENDOR	0x00
 #define	SVPD_ID_TYPE_T10	0x01
 #define	SVPD_ID_TYPE_EUI64	0x02
 #define	SVPD_ID_TYPE_NAA	0x03
 #define	SVPD_ID_TYPE_RELTARG	0x04
 #define	SVPD_ID_TYPE_TPORTGRP	0x05
 #define	SVPD_ID_TYPE_LUNGRP	0x06
 #define	SVPD_ID_TYPE_MD5_LUN_ID	0x07
 #define	SVPD_ID_TYPE_SCSI_NAME	0x08
 #define	SVPD_ID_TYPE_MASK	0x0f
 	u_int8_t	reserved;
 	u_int8_t	length;
 #define	SVPD_DEVICE_ID_DESC_HDR_LEN \
     __offsetof(struct scsi_vpd_id_descriptor, identifier) 
 	u_int8_t	identifier[];
 };
 
 struct scsi_vpd_id_t10
 {
 	u_int8_t	vendor[8];
 	u_int8_t	vendor_spec_id[0];
 };
 
 struct scsi_vpd_id_eui64
 {
 	u_int8_t	ieee_company_id[3];
 	u_int8_t	extension_id[5];
 };
 
 struct scsi_vpd_id_naa_basic
 {
 	uint8_t naa;
 	/* big endian, packed:
 	uint8_t	naa : 4;
 	uint8_t naa_desig : 4;
 	*/
 #define	SVPD_ID_NAA_NAA_SHIFT		4
 #define	SVPD_ID_NAA_IEEE_EXT		0x02
 #define	SVPD_ID_NAA_LOCAL_REG		0x03
 #define	SVPD_ID_NAA_IEEE_REG		0x05
 #define	SVPD_ID_NAA_IEEE_REG_EXT	0x06
 	uint8_t	naa_data[];
 };
 
 struct scsi_vpd_id_naa_ieee_extended_id
 {
 	uint8_t naa;
 	uint8_t vendor_specific_id_a;
 	uint8_t ieee_company_id[3];
 	uint8_t vendor_specific_id_b[4];
 };
 
 struct scsi_vpd_id_naa_local_reg
 {
 	uint8_t naa;
 	uint8_t local_value[7];
 };
 
 struct scsi_vpd_id_naa_ieee_reg
 {
 	uint8_t naa;
 	uint8_t reg_value[7];
 	/* big endian, packed:
 	uint8_t naa_basic : 4;
 	uint8_t ieee_company_id_0 : 4;
 	uint8_t ieee_company_id_1[2];
 	uint8_t ieee_company_id_2 : 4;
 	uint8_t vendor_specific_id_0 : 4;
 	uint8_t vendor_specific_id_1[4];
 	*/
 };
 
 struct scsi_vpd_id_naa_ieee_reg_extended
 {
 	uint8_t naa;
 	uint8_t reg_value[15];
 	/* big endian, packed:
 	uint8_t naa_basic : 4;
 	uint8_t ieee_company_id_0 : 4;
 	uint8_t ieee_company_id_1[2];
 	uint8_t ieee_company_id_2 : 4;
 	uint8_t vendor_specific_id_0 : 4;
 	uint8_t vendor_specific_id_1[4];
 	uint8_t vendor_specific_id_ext[8];
 	*/
 };
 
 struct scsi_vpd_id_rel_trgt_port_id
 {
 	uint8_t obsolete[2];
 	uint8_t rel_trgt_port_id[2];
 };
 
 struct scsi_vpd_id_trgt_port_grp_id
 {
 	uint8_t reserved[2];
 	uint8_t trgt_port_grp[2];
 };
 
 struct scsi_vpd_id_lun_grp_id
 {
 	uint8_t reserved[2];
 	uint8_t log_unit_grp[2];
 };
 
 struct scsi_vpd_id_md5_lun_id
 {
 	uint8_t lun_id[16];
 };
 
 struct scsi_vpd_id_scsi_name
 {
 	uint8_t name_string[256];
 };
 
 struct scsi_service_action_in
 {
 	uint8_t opcode;
 	uint8_t service_action;
 	uint8_t action_dependent[13];
 	uint8_t control;
 };
 
 struct scsi_vpd_extended_inquiry_data
 {
 	uint8_t device;
 	uint8_t page_code;
 #define	SVPD_EXTENDED_INQUIRY_DATA	0x86
-	uint8_t reserved;
-	uint8_t page_length;
+	uint8_t page_length[2];
 	uint8_t flags1;
-#define	SVPD_EID_AM		0xC0
+
+	/* These values are for direct access devices */
+#define	SVPD_EID_AM_MASK	0xC0
+#define	SVPD_EID_AM_DEFER	0x80
+#define	SVPD_EID_AM_IMMED	0x40
+#define	SVPD_EID_AM_UNDEFINED	0x00
+#define	SVPD_EID_AM_RESERVED	0xc0
 #define	SVPD_EID_SPT		0x38
 #define	SVPD_EID_SPT_1		0x00
 #define	SVPD_EID_SPT_12		0x08
 #define	SVPD_EID_SPT_2		0x10
 #define	SVPD_EID_SPT_13		0x18
 #define	SVPD_EID_SPT_3		0x20
 #define	SVPD_EID_SPT_23		0x28
 #define	SVPD_EID_SPT_123	0x38
+
+	/* These values are for sequential access devices */
+#define	SVPD_EID_SA_SPT_LBP	0x08
+
 #define	SVPD_EID_GRD_CHK	0x04
 #define	SVPD_EID_APP_CHK	0x02
 #define	SVPD_EID_REF_CHK	0x01
+
 	uint8_t flags2;
 #define	SVPD_EID_UASK_SUP	0x20
 #define	SVPD_EID_GROUP_SUP	0x10
 #define	SVPD_EID_PRIOR_SUP	0x08
 #define	SVPD_EID_HEADSUP	0x04
 #define	SVPD_EID_ORDSUP		0x02
 #define	SVPD_EID_SIMPSUP	0x01
 	uint8_t flags3;
 #define	SVPD_EID_WU_SUP		0x08
 #define	SVPD_EID_CRD_SUP	0x04
 #define	SVPD_EID_NV_SUP		0x02
 #define	SVPD_EID_V_SUP		0x01
 	uint8_t flags4;
 #define	SVPD_EID_P_I_I_SUP	0x10
 #define	SVPD_EID_LUICLT		0x01
 	uint8_t flags5;
 #define	SVPD_EID_R_SUP		0x10
 #define	SVPD_EID_CBCS		0x01
 	uint8_t flags6;
 #define	SVPD_EID_MULTI_I_T_FW	0x0F
+#define	SVPD_EID_MC_VENDOR_SPEC	0x00
+#define	SVPD_EID_MC_MODE_1	0x01
+#define	SVPD_EID_MC_MODE_2	0x02
+#define	SVPD_EID_MC_MODE_3	0x03
 	uint8_t est[2];
 	uint8_t flags7;
 #define	SVPD_EID_POA_SUP	0x80
 #define	SVPD_EID_HRA_SUP	0x80
 #define	SVPD_EID_VSA_SUP	0x80
 	uint8_t max_sense_length;
 	uint8_t reserved2[50];
 };
 
 struct scsi_vpd_mode_page_policy_descr
 {
 	uint8_t page_code;
 	uint8_t subpage_code;
 	uint8_t policy;
 #define	SVPD_MPP_SHARED		0x00
 #define	SVPD_MPP_PORT		0x01
 #define	SVPD_MPP_I_T		0x03
 #define	SVPD_MPP_MLUS		0x80
 	uint8_t reserved;
 };
 
 struct scsi_vpd_mode_page_policy
 {
 	uint8_t device;
 	uint8_t page_code;
 #define	SVPD_MODE_PAGE_POLICY	0x87
 	uint8_t page_length[2];
 	struct scsi_vpd_mode_page_policy_descr descr[0];
 };
 
 struct scsi_diag_page {
 	uint8_t page_code;
 	uint8_t page_specific_flags;
 	uint8_t length[2];
 	uint8_t params[0];
 };
 
 struct scsi_vpd_port_designation
 {
 	uint8_t reserved[2];
 	uint8_t relative_port_id[2];
 	uint8_t reserved2[2];
 	uint8_t initiator_transportid_length[2];
 	uint8_t initiator_transportid[0];
 };
 
 struct scsi_vpd_port_designation_cont
 {
 	uint8_t reserved[2];
 	uint8_t target_port_descriptors_length[2];
 	struct scsi_vpd_id_descriptor target_port_descriptors[0];
 };
 
 struct scsi_vpd_scsi_ports
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define	SVPD_SCSI_PORTS		0x88
 	u_int8_t page_length[2];
 	struct scsi_vpd_port_designation design[];
 };
 
 /*
  * ATA Information VPD Page based on
  * T10/2126-D Revision 04
  */
 #define SVPD_ATA_INFORMATION		0x89
 
 
 struct scsi_vpd_tpc_descriptor
 {
 	uint8_t desc_type[2];
 	uint8_t desc_length[2];
 	uint8_t parameters[];
 };
 
 struct scsi_vpd_tpc_descriptor_bdrl
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_BDRL			0x0000
 	uint8_t desc_length[2];
 	uint8_t vendor_specific[6];
 	uint8_t maximum_ranges[2];
 	uint8_t maximum_inactivity_timeout[4];
 	uint8_t default_inactivity_timeout[4];
 	uint8_t maximum_token_transfer_size[8];
 	uint8_t optimal_transfer_count[8];
 };
 
 struct scsi_vpd_tpc_descriptor_sc_descr
 {
 	uint8_t opcode;
 	uint8_t sa_length;
 	uint8_t supported_service_actions[0];
 };
 
 struct scsi_vpd_tpc_descriptor_sc
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_SC			0x0001
 	uint8_t desc_length[2];
 	uint8_t list_length;
 	struct scsi_vpd_tpc_descriptor_sc_descr descr[];
 };
 
 struct scsi_vpd_tpc_descriptor_pd
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_PD			0x0004
 	uint8_t desc_length[2];
 	uint8_t reserved[4];
 	uint8_t maximum_cscd_descriptor_count[2];
 	uint8_t maximum_segment_descriptor_count[2];
 	uint8_t maximum_descriptor_list_length[4];
 	uint8_t maximum_inline_data_length[4];
 	uint8_t reserved2[12];
 };
 
 struct scsi_vpd_tpc_descriptor_sd
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_SD			0x0008
 	uint8_t desc_length[2];
 	uint8_t list_length;
 	uint8_t supported_descriptor_codes[];
 };
 
 struct scsi_vpd_tpc_descriptor_sdid
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_SDID			0x000C
 	uint8_t desc_length[2];
 	uint8_t list_length[2];
 	uint8_t supported_descriptor_ids[];
 };
 
 struct scsi_vpd_tpc_descriptor_rtf_block
 {
 	uint8_t type_format;
 #define	SVPD_TPC_RTF_BLOCK			0x00
 	uint8_t reserved;
 	uint8_t desc_length[2];
 	uint8_t reserved2[2];
 	uint8_t optimal_length_granularity[2];
 	uint8_t maximum_bytes[8];
 	uint8_t optimal_bytes[8];
 	uint8_t optimal_bytes_to_token_per_segment[8];
 	uint8_t optimal_bytes_from_token_per_segment[8];
 	uint8_t reserved3[8];
 };
 
 struct scsi_vpd_tpc_descriptor_rtf
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_RTF			0x0106
 	uint8_t desc_length[2];
 	uint8_t remote_tokens;
 	uint8_t reserved[11];
 	uint8_t minimum_token_lifetime[4];
 	uint8_t maximum_token_lifetime[4];
 	uint8_t maximum_token_inactivity_timeout[4];
 	uint8_t reserved2[18];
 	uint8_t type_specific_features_length[2];
 	uint8_t type_specific_features[0];
 };
 
 struct scsi_vpd_tpc_descriptor_srtd
 {
 	uint8_t rod_type[4];
 	uint8_t flags;
 #define	SVPD_TPC_SRTD_TOUT		0x01
 #define	SVPD_TPC_SRTD_TIN		0x02
 #define	SVPD_TPC_SRTD_ECPY		0x80
 	uint8_t reserved;
 	uint8_t preference_indicator[2];
 	uint8_t reserved2[56];
 };
 
 struct scsi_vpd_tpc_descriptor_srt
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_SRT			0x0108
 	uint8_t desc_length[2];
 	uint8_t reserved[2];
 	uint8_t rod_type_descriptors_length[2];
 	uint8_t rod_type_descriptors[0];
 };
 
 struct scsi_vpd_tpc_descriptor_gco
 {
 	uint8_t desc_type[2];
 #define	SVPD_TPC_GCO			0x8001
 	uint8_t desc_length[2];
 	uint8_t total_concurrent_copies[4];
 	uint8_t maximum_identified_concurrent_copies[4];
 	uint8_t maximum_segment_length[4];
 	uint8_t data_segment_granularity;
 	uint8_t inline_data_granularity;
 	uint8_t reserved[18];
 };
 
 struct scsi_vpd_tpc
 {
 	uint8_t device;
 	uint8_t page_code;
 #define	SVPD_SCSI_TPC			0x8F
 	uint8_t page_length[2];
 	struct scsi_vpd_tpc_descriptor descr[];
 };
 
 /*
  * Block Device Characteristics VPD Page based on
  * T10/1799-D Revision 31
  */
 struct scsi_vpd_block_characteristics
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define SVPD_BDC			0xB1
 	u_int8_t page_length[2];
 	u_int8_t medium_rotation_rate[2];
 #define SVPD_BDC_RATE_NOT_REPORTED	0x00
 #define SVPD_BDC_RATE_NON_ROTATING	0x01
 	u_int8_t reserved1;
 	u_int8_t nominal_form_factor;
 #define SVPD_BDC_FORM_NOT_REPORTED	0x00
 #define SVPD_BDC_FORM_5_25INCH		0x01
 #define SVPD_BDC_FORM_3_5INCH		0x02
 #define SVPD_BDC_FORM_2_5INCH		0x03
 #define SVPD_BDC_FORM_1_5INCH		0x04
 #define SVPD_BDC_FORM_LESSTHAN_1_5INCH	0x05
 	u_int8_t reserved2[56];
 };
 
 /*
  * Block Device Characteristics VPD Page
  */
 struct scsi_vpd_block_device_characteristics
 {
 	uint8_t device;
 	uint8_t page_code;
 #define	SVPD_BDC		0xB1
 	uint8_t page_length[2];
 	uint8_t medium_rotation_rate[2];
 #define	SVPD_NOT_REPORTED	0x0000
 #define	SVPD_NON_ROTATING	0x0001
 	uint8_t product_type;
 	uint8_t wab_wac_ff;
 	uint8_t flags;
 #define	SVPD_VBULS		0x01
 #define	SVPD_FUAB		0x02
 #define	SVPD_HAW_ZBC		0x10
 	uint8_t reserved[55];
 };
 
 /*
  * Logical Block Provisioning VPD Page based on
  * T10/1799-D Revision 31
  */
 struct scsi_vpd_logical_block_prov
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define	SVPD_LBP		0xB2
 	u_int8_t page_length[2];
 #define SVPD_LBP_PL_BASIC	0x04
 	u_int8_t threshold_exponent;
 	u_int8_t flags;
 #define SVPD_LBP_UNMAP		0x80
 #define SVPD_LBP_WS16		0x40
 #define SVPD_LBP_WS10		0x20
 #define SVPD_LBP_RZ		0x04
 #define SVPD_LBP_ANC_SUP	0x02
 #define SVPD_LBP_DP		0x01
 	u_int8_t prov_type;
 #define SVPD_LBP_RESOURCE	0x01
 #define SVPD_LBP_THIN		0x02
 	u_int8_t reserved;
 	/*
 	 * Provisioning Group Descriptor can be here if SVPD_LBP_DP is set
 	 * Its size can be determined from page_length - 4
 	 */
 };
 
 /*
  * Block Limits VDP Page based on SBC-4 Revision 2
  */
 struct scsi_vpd_block_limits
 {
 	u_int8_t device;
 	u_int8_t page_code;
 #define	SVPD_BLOCK_LIMITS	0xB0
 	u_int8_t page_length[2];
 #define SVPD_BL_PL_BASIC	0x10
 #define SVPD_BL_PL_TP		0x3C
 	u_int8_t reserved1;
 	u_int8_t max_cmp_write_len;
 	u_int8_t opt_txfer_len_grain[2];
 	u_int8_t max_txfer_len[4];
 	u_int8_t opt_txfer_len[4];
 	u_int8_t max_prefetch[4];
 	u_int8_t max_unmap_lba_cnt[4];
 	u_int8_t max_unmap_blk_cnt[4];
 	u_int8_t opt_unmap_grain[4];
 	u_int8_t unmap_grain_align[4];
 	u_int8_t max_write_same_length[8];
 	u_int8_t max_atomic_transfer_length[4];
 	u_int8_t atomic_alignment[4];
 	u_int8_t atomic_transfer_length_granularity[4];
 	u_int8_t reserved2[8];
 };
 
 struct scsi_read_capacity
 {
 	u_int8_t opcode;
 	u_int8_t byte2;
 #define	SRC_RELADR	0x01
 	u_int8_t addr[4];
 	u_int8_t unused[2];
 	u_int8_t pmi;
 #define	SRC_PMI		0x01
 	u_int8_t control;
 };
 
 struct scsi_read_capacity_16
 {
 	uint8_t opcode;
 #define	SRC16_SERVICE_ACTION	0x10
 	uint8_t service_action;
 	uint8_t addr[8];
 	uint8_t alloc_len[4];
 #define	SRC16_PMI		0x01
 #define	SRC16_RELADR		0x02
 	uint8_t reladr;
 	uint8_t control;
 };
 
 struct scsi_read_capacity_data
 {
 	u_int8_t addr[4];
 	u_int8_t length[4];
 };
 
 struct scsi_read_capacity_data_long
 {
 	uint8_t addr[8];
 	uint8_t length[4];
 #define	SRC16_PROT_EN		0x01
 #define	SRC16_P_TYPE		0x0e
 #define	SRC16_PTYPE_1		0x00
 #define	SRC16_PTYPE_2		0x02
 #define	SRC16_PTYPE_3		0x04
 	uint8_t prot;
 #define	SRC16_LBPPBE		0x0f
 #define	SRC16_PI_EXPONENT	0xf0
 #define	SRC16_PI_EXPONENT_SHIFT	4
 	uint8_t prot_lbppbe;
 #define	SRC16_LALBA		0x3f
 #define	SRC16_LBPRZ		0x40
 #define	SRC16_LBPME		0x80
 /*
  * Alternate versions of these macros that are intended for use on a 16-bit
  * version of the lalba_lbp field instead of the array of 2 8 bit numbers.
  */
 #define	SRC16_LALBA_A		0x3fff
 #define	SRC16_LBPRZ_A		0x4000
 #define	SRC16_LBPME_A		0x8000
 	uint8_t lalba_lbp[2];
 	uint8_t	reserved[16];
 };
 
 struct scsi_get_lba_status
 {
 	uint8_t opcode;
 #define	SGLS_SERVICE_ACTION	0x12
 	uint8_t service_action;
 	uint8_t addr[8];
 	uint8_t alloc_len[4];
 	uint8_t reserved;
 	uint8_t control;
 };
 
 struct scsi_get_lba_status_data_descr
 {
 	uint8_t addr[8];
 	uint8_t length[4];
 	uint8_t status;
 	uint8_t reserved[3];
 };
 
 struct scsi_get_lba_status_data
 {
 	uint8_t length[4];
 	uint8_t reserved[4];
 	struct scsi_get_lba_status_data_descr descr[];
 };
 
 struct scsi_report_luns
 {
 	uint8_t opcode;
 	uint8_t reserved1;
 #define	RPL_REPORT_DEFAULT	0x00
 #define	RPL_REPORT_WELLKNOWN	0x01
 #define	RPL_REPORT_ALL		0x02
 	uint8_t select_report;
 	uint8_t reserved2[3];
 	uint8_t length[4];
 	uint8_t reserved3;
 	uint8_t control;
 };
 
 struct scsi_report_luns_lundata {
 	uint8_t lundata[8];
 #define	RPL_LUNDATA_PERIPH_BUS_MASK	0x3f
 #define	RPL_LUNDATA_FLAT_LUN_MASK	0x3f
 #define	RPL_LUNDATA_FLAT_LUN_BITS	0x06
 #define	RPL_LUNDATA_LUN_TARG_MASK	0x3f
 #define	RPL_LUNDATA_LUN_BUS_MASK	0xe0
 #define	RPL_LUNDATA_LUN_LUN_MASK	0x1f
 #define	RPL_LUNDATA_EXT_LEN_MASK	0x30
 #define	RPL_LUNDATA_EXT_EAM_MASK	0x0f
 #define	RPL_LUNDATA_EXT_EAM_WK		0x01
 #define	RPL_LUNDATA_EXT_EAM_NOT_SPEC	0x0f
 #define	RPL_LUNDATA_ATYP_MASK	0xc0	/* MBZ for type 0 lun */
 #define	RPL_LUNDATA_ATYP_PERIPH	0x00
 #define	RPL_LUNDATA_ATYP_FLAT	0x40
 #define	RPL_LUNDATA_ATYP_LUN	0x80
 #define	RPL_LUNDATA_ATYP_EXTLUN	0xc0
 };
 
 struct scsi_report_luns_data {
 	u_int8_t length[4];	/* length of LUN inventory, in bytes */
 	u_int8_t reserved[4];	/* unused */
 	/*
 	 * LUN inventory- we only support the type zero form for now.
 	 */
 	struct scsi_report_luns_lundata luns[0];
 };
 
 struct scsi_target_group
 {
 	uint8_t opcode;
 	uint8_t service_action;
 #define	STG_PDF_MASK		0xe0
 #define	STG_PDF_LENGTH		0x00
 #define	STG_PDF_EXTENDED	0x20
 	uint8_t reserved1[4];
 	uint8_t length[4];
 	uint8_t reserved2;
 	uint8_t control;
 };
 
 struct scsi_target_port_descriptor {
 	uint8_t	reserved[2];
 	uint8_t	relative_target_port_identifier[2];
 	uint8_t desc_list[];
 };
 
 struct scsi_target_port_group_descriptor {
 	uint8_t	pref_state;
 #define	TPG_PRIMARY				0x80
 #define	TPG_ASYMMETRIC_ACCESS_STATE_MASK	0xf
 #define	TPG_ASYMMETRIC_ACCESS_OPTIMIZED		0x0
 #define	TPG_ASYMMETRIC_ACCESS_NONOPTIMIZED	0x1
 #define	TPG_ASYMMETRIC_ACCESS_STANDBY		0x2
 #define	TPG_ASYMMETRIC_ACCESS_UNAVAILABLE	0x3
 #define	TPG_ASYMMETRIC_ACCESS_LBA_DEPENDENT	0x4
 #define	TPG_ASYMMETRIC_ACCESS_OFFLINE		0xE
 #define	TPG_ASYMMETRIC_ACCESS_TRANSITIONING	0xF
 	uint8_t support;
 #define	TPG_AO_SUP	0x01
 #define	TPG_AN_SUP	0x02
 #define	TPG_S_SUP	0x04
 #define	TPG_U_SUP	0x08
 #define	TPG_LBD_SUP	0x10
 #define	TPG_O_SUP	0x40
 #define	TPG_T_SUP	0x80
 	uint8_t target_port_group[2];
 	uint8_t reserved;
 	uint8_t status;
 #define TPG_UNAVLBL      0
 #define TPG_SET_BY_STPG  0x01
 #define TPG_IMPLICIT     0x02
 	uint8_t vendor_specific;
 	uint8_t	target_port_count;
 	struct scsi_target_port_descriptor descriptors[];
 };
 
 struct scsi_target_group_data {
 	uint8_t length[4];	/* length of returned data, in bytes */
 	struct scsi_target_port_group_descriptor groups[];
 };
 
 struct scsi_target_group_data_extended {
 	uint8_t length[4];	/* length of returned data, in bytes */
 	uint8_t format_type;	/* STG_PDF_LENGTH or STG_PDF_EXTENDED */
 	uint8_t	implicit_transition_time;
 	uint8_t reserved[2];
 	struct scsi_target_port_group_descriptor groups[];
 };
 
 
 typedef enum {
 	SSD_TYPE_NONE,
 	SSD_TYPE_FIXED,
 	SSD_TYPE_DESC
 } scsi_sense_data_type;
 
 typedef enum {
 	SSD_ELEM_NONE,
 	SSD_ELEM_SKIP,
 	SSD_ELEM_DESC,
 	SSD_ELEM_SKS,
 	SSD_ELEM_COMMAND,
 	SSD_ELEM_INFO,
 	SSD_ELEM_FRU,
 	SSD_ELEM_STREAM,
 	SSD_ELEM_MAX
 } scsi_sense_elem_type;
 
 
 struct scsi_sense_data
 {
 	uint8_t error_code;
 	/*
 	 * SPC-4 says that the maximum length of sense data is 252 bytes.
 	 * So this structure is exactly 252 bytes log.
 	 */
 #define	SSD_FULL_SIZE 252
 	uint8_t sense_buf[SSD_FULL_SIZE - 1];
 	/*
 	 * XXX KDM is this still a reasonable minimum size?
 	 */
 #define	SSD_MIN_SIZE 18
 	/*
 	 * Maximum value for the extra_len field in the sense data.
 	 */
 #define	SSD_EXTRA_MAX 244
 };
 
 /*
  * Fixed format sense data.
  */
 struct scsi_sense_data_fixed
 {
 	u_int8_t error_code;
 #define	SSD_ERRCODE			0x7F
 #define		SSD_CURRENT_ERROR	0x70
 #define		SSD_DEFERRED_ERROR	0x71
 #define	SSD_ERRCODE_VALID	0x80	
 	u_int8_t segment;
 	u_int8_t flags;
 #define	SSD_KEY				0x0F
 #define		SSD_KEY_NO_SENSE	0x00
 #define		SSD_KEY_RECOVERED_ERROR	0x01
 #define		SSD_KEY_NOT_READY	0x02
 #define		SSD_KEY_MEDIUM_ERROR	0x03
 #define		SSD_KEY_HARDWARE_ERROR	0x04
 #define		SSD_KEY_ILLEGAL_REQUEST	0x05
 #define		SSD_KEY_UNIT_ATTENTION	0x06
 #define		SSD_KEY_DATA_PROTECT	0x07
 #define		SSD_KEY_BLANK_CHECK	0x08
 #define		SSD_KEY_Vendor_Specific	0x09
 #define		SSD_KEY_COPY_ABORTED	0x0a
 #define		SSD_KEY_ABORTED_COMMAND	0x0b		
 #define		SSD_KEY_EQUAL		0x0c
 #define		SSD_KEY_VOLUME_OVERFLOW	0x0d
 #define		SSD_KEY_MISCOMPARE	0x0e
 #define		SSD_KEY_COMPLETED	0x0f			
 #define	SSD_ILI		0x20
 #define	SSD_EOM		0x40
 #define	SSD_FILEMARK	0x80
 	u_int8_t info[4];
 	u_int8_t extra_len;
 	u_int8_t cmd_spec_info[4];
 	u_int8_t add_sense_code;
 	u_int8_t add_sense_code_qual;
 	u_int8_t fru;
 	u_int8_t sense_key_spec[3];
 #define	SSD_SCS_VALID		0x80
 #define	SSD_FIELDPTR_CMD	0x40
 #define	SSD_BITPTR_VALID	0x08
 #define	SSD_BITPTR_VALUE	0x07
 	u_int8_t extra_bytes[14];
 #define	SSD_FIXED_IS_PRESENT(sense, length, field) 			\
 	((length >= (offsetof(struct scsi_sense_data_fixed, field) +	\
 	sizeof(sense->field))) ? 1 :0)
 #define	SSD_FIXED_IS_FILLED(sense, field) 				\
 	((((offsetof(struct scsi_sense_data_fixed, field) +		\
 	sizeof(sense->field)) -						\
 	(offsetof(struct scsi_sense_data_fixed, extra_len) +		\
 	sizeof(sense->extra_len))) <= sense->extra_len) ? 1 : 0)
 };
 
 /*
  * Descriptor format sense data definitions.
  * Introduced in SPC-3.
  */
 struct scsi_sense_data_desc 
 {
 	uint8_t	error_code;
 #define	SSD_DESC_CURRENT_ERROR	0x72
 #define	SSD_DESC_DEFERRED_ERROR	0x73
 	uint8_t sense_key;
 	uint8_t	add_sense_code;
 	uint8_t	add_sense_code_qual;
 	uint8_t	reserved[3];
 	/*
 	 * Note that SPC-4, section 4.5.2.1 says that the extra_len field
 	 * must be less than or equal to 244.
 	 */
 	uint8_t	extra_len;
 	uint8_t	sense_desc[0];
 #define	SSD_DESC_IS_PRESENT(sense, length, field) 			\
 	((length >= (offsetof(struct scsi_sense_data_desc, field) +	\
 	sizeof(sense->field))) ? 1 :0)
 };
 
 struct scsi_sense_desc_header
 {
 	uint8_t desc_type;
 	uint8_t length;
 };
 /*
  * The information provide in the Information descriptor is device type or
  * command specific information, and defined in a command standard.
  *
  * Note that any changes to the field names or positions in this structure,
  * even reserved fields, should be accompanied by an examination of the
  * code in ctl_set_sense() that uses them.
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_info
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_INFO	0x00
 	uint8_t	length;
 	uint8_t	byte2;
 #define	SSD_INFO_VALID	0x80
 	uint8_t	reserved;
 	uint8_t	info[8];
 };
 
 /*
  * Command-specific information depends on the command for which the
  * reported condition occured.
  *
  * Note that any changes to the field names or positions in this structure,
  * even reserved fields, should be accompanied by an examination of the
  * code in ctl_set_sense() that uses them.
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_command
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_COMMAND	0x01
 	uint8_t	length;
 	uint8_t	reserved[2];
 	uint8_t	command_info[8];
 };
 
 /*
  * Sense key specific descriptor.  The sense key specific data format
  * depends on the sense key in question.
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_sks
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_SKS		0x02
 	uint8_t	length;
 	uint8_t reserved1[2];
 	uint8_t	sense_key_spec[3];
 #define	SSD_SKS_VALID		0x80
 	uint8_t reserved2;
 };
 
 /*
  * This is used for the Illegal Request sense key (0x05) only.
  */
 struct scsi_sense_sks_field
 {
 	uint8_t	byte0;
 #define	SSD_SKS_FIELD_VALID	0x80
 #define	SSD_SKS_FIELD_CMD	0x40
 #define	SSD_SKS_BPV		0x08
 #define	SSD_SKS_BIT_VALUE	0x07
 	uint8_t	field[2];
 };
 
 
 /* 
  * This is used for the Hardware Error (0x04), Medium Error (0x03) and
  * Recovered Error (0x01) sense keys.
  */
 struct scsi_sense_sks_retry
 {
 	uint8_t byte0;
 #define	SSD_SKS_RETRY_VALID	0x80
 	uint8_t actual_retry_count[2];
 };
 
 /*
  * Used with the NO Sense (0x00) or Not Ready (0x02) sense keys.
  */
 struct scsi_sense_sks_progress
 {
 	uint8_t byte0;
 #define	SSD_SKS_PROGRESS_VALID	0x80
 	uint8_t progress[2];
 #define	SSD_SKS_PROGRESS_DENOM	0x10000
 };
 
 /*
  * Used with the Copy Aborted (0x0a) sense key.
  */
 struct scsi_sense_sks_segment
 {
 	uint8_t byte0;
 #define	SSD_SKS_SEGMENT_VALID	0x80
 #define	SSD_SKS_SEGMENT_SD	0x20
 #define	SSD_SKS_SEGMENT_BPV	0x08
 #define	SSD_SKS_SEGMENT_BITPTR	0x07
 	uint8_t field[2];
 };
 
 /*
  * Used with the Unit Attention (0x06) sense key.
  *
  * This is currently used to indicate that the unit attention condition
  * queue has overflowed (when the overflow bit is set).
  */
 struct scsi_sense_sks_overflow
 {
 	uint8_t byte0;
 #define	SSD_SKS_OVERFLOW_VALID	0x80
 #define	SSD_SKS_OVERFLOW_SET	0x01
 	uint8_t	reserved[2];
 };
 
 /*
  * This specifies which component is associated with the sense data.  There
  * is no standard meaning for the fru value.
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_fru
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_FRU		0x03
 	uint8_t	length;
 	uint8_t reserved;
 	uint8_t fru;
 };
 
 /*
  * Used for Stream commands, defined in SSC-4.
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
  
 struct scsi_sense_stream
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_STREAM		0x04
 	uint8_t	length;
 	uint8_t	reserved;
 	uint8_t	byte3;
 #define	SSD_DESC_STREAM_FM	0x80
 #define	SSD_DESC_STREAM_EOM	0x40
 #define	SSD_DESC_STREAM_ILI	0x20
 };
 
 /*
  * Used for Block commands, defined in SBC-3.
  *
  * This is currently (as of SBC-3) only used for the Incorrect Length
  * Indication (ILI) bit, which says that the data length requested in the
  * READ LONG or WRITE LONG command did not match the length of the logical
  * block.
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_block
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_BLOCK		0x05
 	uint8_t	length;
 	uint8_t	reserved;
 	uint8_t	byte3;
 #define	SSD_DESC_BLOCK_ILI	0x20
 };
 
 /*
  * Used for Object-Based Storage Devices (OSD-3).
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_osd_objid
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_OSD_OBJID	0x06
 	uint8_t	length;
 	uint8_t	reserved[6];
 	/*
 	 * XXX KDM provide the bit definitions here?  There are a lot of
 	 * them, and we don't have an OSD driver yet.
 	 */
 	uint8_t	not_init_cmds[4];
 	uint8_t	completed_cmds[4];
 	uint8_t	partition_id[8];
 	uint8_t	object_id[8];
 };
 
 /*
  * Used for Object-Based Storage Devices (OSD-3).
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_osd_integrity
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_OSD_INTEGRITY	0x07
 	uint8_t	length;
 	uint8_t	integ_check_val[32];
 };
 
 /*
  * Used for Object-Based Storage Devices (OSD-3).
  *
  * Maximum descriptors allowed: 1 (as of SPC-4)
  */
 struct scsi_sense_osd_attr_id
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_OSD_ATTR_ID	0x08
 	uint8_t	length;
 	uint8_t	reserved[2];
 	uint8_t	attr_desc[0];
 };
 
 /*
  * Used with Sense keys No Sense (0x00) and Not Ready (0x02).
  *
  * Maximum descriptors allowed: 32 (as of SPC-4)
  */
 struct scsi_sense_progress
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_PROGRESS	0x0a
 	uint8_t	length;
 	uint8_t	sense_key;
 	uint8_t	add_sense_code;
 	uint8_t	add_sense_code_qual;
 	uint8_t reserved;
 	uint8_t	progress[2];
 };
 
 /*
  * This is typically forwarded as the result of an EXTENDED COPY command.
  *
  * Maximum descriptors allowed: 2 (as of SPC-4)
  */
 struct scsi_sense_forwarded
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_FORWARDED	0x0c
 	uint8_t	length;
 	uint8_t	byte2;
 #define	SSD_FORWARDED_FSDT	0x80
 #define	SSD_FORWARDED_SDS_MASK	0x0f
 #define	SSD_FORWARDED_SDS_UNK	0x00
 #define	SSD_FORWARDED_SDS_EXSRC	0x01
 #define	SSD_FORWARDED_SDS_EXDST	0x02
 };
 
 /*
  * Vendor-specific sense descriptor.  The desc_type field will be in the
  * range bewteen MIN and MAX inclusive.
  */
 struct scsi_sense_vendor
 {
 	uint8_t	desc_type;
 #define	SSD_DESC_VENDOR_MIN	0x80
 #define	SSD_DESC_VENDOR_MAX	0xff
 	uint8_t length;
 	uint8_t	data[0];
 };
 
 struct scsi_mode_header_6
 {
 	u_int8_t data_length;	/* Sense data length */
 	u_int8_t medium_type;
 	u_int8_t dev_spec;
 	u_int8_t blk_desc_len;
 };
 
 struct scsi_mode_header_10
 {
 	u_int8_t data_length[2];/* Sense data length */
 	u_int8_t medium_type;
 	u_int8_t dev_spec;
 	u_int8_t unused[2];
 	u_int8_t blk_desc_len[2];
 };
 
 struct scsi_mode_page_header
 {
 	u_int8_t page_code;
 #define	SMPH_PS		0x80
 #define	SMPH_SPF	0x40
 #define	SMPH_PC_MASK	0x3f
 	u_int8_t page_length;
 };
 
 struct scsi_mode_page_header_sp
 {
 	uint8_t page_code;
 	uint8_t subpage;
 	uint8_t page_length[2];
 };
 
 
 struct scsi_mode_blk_desc
 {
 	u_int8_t density;
 	u_int8_t nblocks[3];
 	u_int8_t reserved;
 	u_int8_t blklen[3];
 };
 
 #define	SCSI_DEFAULT_DENSITY	0x00	/* use 'default' density */
 #define	SCSI_SAME_DENSITY	0x7f	/* use 'same' density- >= SCSI-2 only */
 
 
 /*
  * Status Byte
  */
 #define	SCSI_STATUS_OK			0x00
 #define	SCSI_STATUS_CHECK_COND		0x02
 #define	SCSI_STATUS_COND_MET		0x04
 #define	SCSI_STATUS_BUSY		0x08
 #define	SCSI_STATUS_INTERMED		0x10
 #define	SCSI_STATUS_INTERMED_COND_MET	0x14
 #define	SCSI_STATUS_RESERV_CONFLICT	0x18
 #define	SCSI_STATUS_CMD_TERMINATED	0x22	/* Obsolete in SAM-2 */
 #define	SCSI_STATUS_QUEUE_FULL		0x28
 #define	SCSI_STATUS_ACA_ACTIVE		0x30
 #define	SCSI_STATUS_TASK_ABORTED	0x40
 
 struct scsi_inquiry_pattern {
 	u_int8_t   type;
 	u_int8_t   media_type;
 #define	SIP_MEDIA_REMOVABLE	0x01
 #define	SIP_MEDIA_FIXED		0x02
 	const char *vendor;
 	const char *product;
 	const char *revision;
 }; 
 
 struct scsi_static_inquiry_pattern {
 	u_int8_t   type;
 	u_int8_t   media_type;
 	char       vendor[SID_VENDOR_SIZE+1];
 	char       product[SID_PRODUCT_SIZE+1];
 	char       revision[SID_REVISION_SIZE+1];
 };
 
 struct scsi_sense_quirk_entry {
 	struct scsi_inquiry_pattern	inq_pat;
 	int				num_sense_keys;
 	int				num_ascs;
 	struct sense_key_table_entry	*sense_key_info;
 	struct asc_table_entry		*asc_info;
 };
 
 struct sense_key_table_entry {
 	u_int8_t    sense_key;
 	u_int32_t   action;
 	const char *desc;
 };
 
 struct asc_table_entry {
 	u_int8_t    asc;
 	u_int8_t    ascq;
 	u_int32_t   action;
 	const char *desc;
 };
 
 struct op_table_entry {
 	u_int8_t    opcode;
 	u_int32_t   opmask;
 	const char  *desc;
 };
 
 struct scsi_op_quirk_entry {
 	struct scsi_inquiry_pattern	inq_pat;
 	int				num_ops;
 	struct op_table_entry		*op_table;
 };
 
 typedef enum {
 	SSS_FLAG_NONE		= 0x00,
 	SSS_FLAG_PRINT_COMMAND	= 0x01
 } scsi_sense_string_flags;
 
 struct scsi_nv {
 	const char *name;
 	uint64_t value;
 };
 
 typedef enum {
 	SCSI_NV_FOUND,
 	SCSI_NV_AMBIGUOUS,
 	SCSI_NV_NOT_FOUND
 } scsi_nv_status;
 
 typedef enum {
 	SCSI_NV_FLAG_NONE	= 0x00,
 	SCSI_NV_FLAG_IG_CASE	= 0x01	/* Case insensitive comparison */
 } scsi_nv_flags;
 
 struct ccb_scsiio;
 struct cam_periph;
 union  ccb;
 #ifndef _KERNEL
 struct cam_device;
 #endif
 
 extern const char *scsi_sense_key_text[];
 
 struct sbuf;
 
 __BEGIN_DECLS
 void scsi_sense_desc(int sense_key, int asc, int ascq,
 		     struct scsi_inquiry_data *inq_data,
 		     const char **sense_key_desc, const char **asc_desc);
 scsi_sense_action scsi_error_action(struct ccb_scsiio* csio,
 				    struct scsi_inquiry_data *inq_data,
 				    u_int32_t sense_flags);
 const char *	scsi_status_string(struct ccb_scsiio *csio);
 
 void scsi_desc_iterate(struct scsi_sense_data_desc *sense, u_int sense_len,
 		       int (*iter_func)(struct scsi_sense_data_desc *sense,
 					u_int, struct scsi_sense_desc_header *,
 					void *), void *arg);
 uint8_t *scsi_find_desc(struct scsi_sense_data_desc *sense, u_int sense_len,
 			uint8_t desc_type);
 void scsi_set_sense_data(struct scsi_sense_data *sense_data, 
 			 scsi_sense_data_type sense_format, int current_error,
 			 int sense_key, int asc, int ascq, ...) ;
 void scsi_set_sense_data_va(struct scsi_sense_data *sense_data,
 			    scsi_sense_data_type sense_format,
 			    int current_error, int sense_key, int asc,
 			    int ascq, va_list ap);
 int scsi_get_sense_info(struct scsi_sense_data *sense_data, u_int sense_len,
 			uint8_t info_type, uint64_t *info,
 			int64_t *signed_info);
 int scsi_get_sks(struct scsi_sense_data *sense_data, u_int sense_len,
 		 uint8_t *sks);
 int scsi_get_block_info(struct scsi_sense_data *sense_data, u_int sense_len,
 			struct scsi_inquiry_data *inq_data,
 			uint8_t *block_bits);
 int scsi_get_stream_info(struct scsi_sense_data *sense_data, u_int sense_len,
 			 struct scsi_inquiry_data *inq_data,
 			 uint8_t *stream_bits);
 void scsi_info_sbuf(struct sbuf *sb, uint8_t *cdb, int cdb_len,
 		    struct scsi_inquiry_data *inq_data, uint64_t info);
 void scsi_command_sbuf(struct sbuf *sb, uint8_t *cdb, int cdb_len,
 		       struct scsi_inquiry_data *inq_data, uint64_t csi);
 void scsi_progress_sbuf(struct sbuf *sb, uint16_t progress);
 int scsi_sks_sbuf(struct sbuf *sb, int sense_key, uint8_t *sks);
 void scsi_fru_sbuf(struct sbuf *sb, uint64_t fru);
 void scsi_stream_sbuf(struct sbuf *sb, uint8_t stream_bits, uint64_t info);
 void scsi_block_sbuf(struct sbuf *sb, uint8_t block_bits, uint64_t info);
 void scsi_sense_info_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			  u_int sense_len, uint8_t *cdb, int cdb_len,
 			  struct scsi_inquiry_data *inq_data,
 			  struct scsi_sense_desc_header *header);
 
 void scsi_sense_command_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			     u_int sense_len, uint8_t *cdb, int cdb_len,
 			     struct scsi_inquiry_data *inq_data,
 			     struct scsi_sense_desc_header *header);
 void scsi_sense_sks_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			 u_int sense_len, uint8_t *cdb, int cdb_len,
 			 struct scsi_inquiry_data *inq_data,
 			 struct scsi_sense_desc_header *header);
 void scsi_sense_fru_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			 u_int sense_len, uint8_t *cdb, int cdb_len,
 			 struct scsi_inquiry_data *inq_data,
 			 struct scsi_sense_desc_header *header);
 void scsi_sense_stream_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			    u_int sense_len, uint8_t *cdb, int cdb_len,
 			    struct scsi_inquiry_data *inq_data,
 			    struct scsi_sense_desc_header *header);
 void scsi_sense_block_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			   u_int sense_len, uint8_t *cdb, int cdb_len,
 			   struct scsi_inquiry_data *inq_data,
 			   struct scsi_sense_desc_header *header);
 void scsi_sense_progress_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			      u_int sense_len, uint8_t *cdb, int cdb_len,
 			      struct scsi_inquiry_data *inq_data,
 			      struct scsi_sense_desc_header *header);
 void scsi_sense_generic_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			     u_int sense_len, uint8_t *cdb, int cdb_len,
 			     struct scsi_inquiry_data *inq_data,
 			     struct scsi_sense_desc_header *header);
 void scsi_sense_desc_sbuf(struct sbuf *sb, struct scsi_sense_data *sense,
 			  u_int sense_len, uint8_t *cdb, int cdb_len,
 			  struct scsi_inquiry_data *inq_data,
 			  struct scsi_sense_desc_header *header);
 scsi_sense_data_type scsi_sense_type(struct scsi_sense_data *sense_data);
 
 void scsi_sense_only_sbuf(struct scsi_sense_data *sense, u_int sense_len,
 			  struct sbuf *sb, char *path_str,
 			  struct scsi_inquiry_data *inq_data, uint8_t *cdb,
 			  int cdb_len);
 
 #ifdef _KERNEL
 int		scsi_command_string(struct ccb_scsiio *csio, struct sbuf *sb);
 int		scsi_sense_sbuf(struct ccb_scsiio *csio, struct sbuf *sb,
 				scsi_sense_string_flags flags);
 char *		scsi_sense_string(struct ccb_scsiio *csio,
 				  char *str, int str_len);
 void		scsi_sense_print(struct ccb_scsiio *csio);
 int 		scsi_vpd_supported_page(struct cam_periph *periph,
 					uint8_t page_id);
 #else /* _KERNEL */
 int		scsi_command_string(struct cam_device *device,
 				    struct ccb_scsiio *csio, struct sbuf *sb);
 int		scsi_sense_sbuf(struct cam_device *device, 
 				struct ccb_scsiio *csio, struct sbuf *sb,
 				scsi_sense_string_flags flags);
 char *		scsi_sense_string(struct cam_device *device, 
 				  struct ccb_scsiio *csio,
 				  char *str, int str_len);
 void		scsi_sense_print(struct cam_device *device, 
 				 struct ccb_scsiio *csio, FILE *ofile);
 #endif /* _KERNEL */
 
 const char *	scsi_op_desc(u_int16_t opcode, 
 			     struct scsi_inquiry_data *inq_data);
 char *		scsi_cdb_string(u_int8_t *cdb_ptr, char *cdb_string,
 				size_t len);
 
 void		scsi_print_inquiry(struct scsi_inquiry_data *inq_data);
 void		scsi_print_inquiry_short(struct scsi_inquiry_data *inq_data);
 
 u_int		scsi_calc_syncsrate(u_int period_factor);
 u_int		scsi_calc_syncparam(u_int period);
 
 typedef int	(*scsi_devid_checkfn_t)(uint8_t *);
 int		scsi_devid_is_naa_ieee_reg(uint8_t *bufp);
 int		scsi_devid_is_sas_target(uint8_t *bufp);
 int		scsi_devid_is_lun_eui64(uint8_t *bufp);
 int		scsi_devid_is_lun_naa(uint8_t *bufp);
 int		scsi_devid_is_lun_name(uint8_t *bufp);
 int		scsi_devid_is_lun_t10(uint8_t *bufp);
 struct scsi_vpd_id_descriptor *
 		scsi_get_devid(struct scsi_vpd_device_id *id, uint32_t len,
 			       scsi_devid_checkfn_t ck_fn);
 struct scsi_vpd_id_descriptor *
 		scsi_get_devid_desc(struct scsi_vpd_id_descriptor *desc, uint32_t len,
 			       scsi_devid_checkfn_t ck_fn);
 
 int		scsi_transportid_sbuf(struct sbuf *sb,
 				      struct scsi_transportid_header *hdr,
 				      uint32_t valid_len);
 
 const char *	scsi_nv_to_str(struct scsi_nv *table, int num_table_entries,
 			       uint64_t value);
 
 scsi_nv_status	scsi_get_nv(struct scsi_nv *table, int num_table_entries,
 			    char *name, int *table_entry, scsi_nv_flags flags);
 
 int	scsi_parse_transportid_64bit(int proto_id, char *id_str,
 				     struct scsi_transportid_header **hdr,
 				     unsigned int *alloc_len,
 #ifdef _KERNEL
 				     struct malloc_type *type, int flags,
 #endif
 				     char *error_str, int error_str_len);
 
 int	scsi_parse_transportid_spi(char *id_str,
 				   struct scsi_transportid_header **hdr,
 				   unsigned int *alloc_len,
 #ifdef _KERNEL
 				   struct malloc_type *type, int flags,
 #endif
 				   char *error_str, int error_str_len);
 
 int	scsi_parse_transportid_rdma(char *id_str,
 				    struct scsi_transportid_header **hdr,
 				    unsigned int *alloc_len,
 #ifdef _KERNEL
 				    struct malloc_type *type, int flags,
 #endif
 				    char *error_str, int error_str_len);
 
 int	scsi_parse_transportid_iscsi(char *id_str,
 				     struct scsi_transportid_header **hdr,
 				     unsigned int *alloc_len,
 #ifdef _KERNEL
 				     struct malloc_type *type, int flags,
 #endif
 				     char *error_str,int error_str_len);
 
 int	scsi_parse_transportid_sop(char *id_str,
 				   struct scsi_transportid_header **hdr,
 				   unsigned int *alloc_len,
 #ifdef _KERNEL
 				   struct malloc_type *type, int flags,
 #endif
 				   char *error_str,int error_str_len);
 
 int	scsi_parse_transportid(char *transportid_str,
 			       struct scsi_transportid_header **hdr,
 			       unsigned int *alloc_len,
 #ifdef _KERNEL
 			       struct malloc_type *type, int flags,
 #endif
 			       char *error_str, int error_str_len);
 
 void		scsi_test_unit_ready(struct ccb_scsiio *csio, u_int32_t retries,
 				     void (*cbfcnp)(struct cam_periph *, 
 						    union ccb *),
 				     u_int8_t tag_action, 
 				     u_int8_t sense_len, u_int32_t timeout);
 
 void		scsi_request_sense(struct ccb_scsiio *csio, u_int32_t retries,
 				   void (*cbfcnp)(struct cam_periph *, 
 						  union ccb *),
 				   void *data_ptr, u_int8_t dxfer_len,
 				   u_int8_t tag_action, u_int8_t sense_len,
 				   u_int32_t timeout);
 
 void		scsi_inquiry(struct ccb_scsiio *csio, u_int32_t retries,
 			     void (*cbfcnp)(struct cam_periph *, union ccb *),
 			     u_int8_t tag_action, u_int8_t *inq_buf, 
 			     u_int32_t inq_len, int evpd, u_int8_t page_code,
 			     u_int8_t sense_len, u_int32_t timeout);
 
 void		scsi_mode_sense(struct ccb_scsiio *csio, u_int32_t retries,
 				void (*cbfcnp)(struct cam_periph *,
 					       union ccb *),
 				u_int8_t tag_action, int dbd,
 				u_int8_t page_code, u_int8_t page,
 				u_int8_t *param_buf, u_int32_t param_len,
 				u_int8_t sense_len, u_int32_t timeout);
 
 void		scsi_mode_sense_len(struct ccb_scsiio *csio, u_int32_t retries,
 				    void (*cbfcnp)(struct cam_periph *,
 						   union ccb *),
 				    u_int8_t tag_action, int dbd,
 				    u_int8_t page_code, u_int8_t page,
 				    u_int8_t *param_buf, u_int32_t param_len,
 				    int minimum_cmd_size, u_int8_t sense_len,
 				    u_int32_t timeout);
 
 void		scsi_mode_select(struct ccb_scsiio *csio, u_int32_t retries,
 				 void (*cbfcnp)(struct cam_periph *,
 						union ccb *),
 				 u_int8_t tag_action, int scsi_page_fmt,
 				 int save_pages, u_int8_t *param_buf,
 				 u_int32_t param_len, u_int8_t sense_len,
 				 u_int32_t timeout);
 
 void		scsi_mode_select_len(struct ccb_scsiio *csio, u_int32_t retries,
 				     void (*cbfcnp)(struct cam_periph *,
 						    union ccb *),
 				     u_int8_t tag_action, int scsi_page_fmt,
 				     int save_pages, u_int8_t *param_buf,
 				     u_int32_t param_len, int minimum_cmd_size,
 				     u_int8_t sense_len, u_int32_t timeout);
 
 void		scsi_log_sense(struct ccb_scsiio *csio, u_int32_t retries,
 			       void (*cbfcnp)(struct cam_periph *, union ccb *),
 			       u_int8_t tag_action, u_int8_t page_code,
 			       u_int8_t page, int save_pages, int ppc,
 			       u_int32_t paramptr, u_int8_t *param_buf,
 			       u_int32_t param_len, u_int8_t sense_len,
 			       u_int32_t timeout);
 
 void		scsi_log_select(struct ccb_scsiio *csio, u_int32_t retries,
 				void (*cbfcnp)(struct cam_periph *,
 				union ccb *), u_int8_t tag_action,
 				u_int8_t page_code, int save_pages,
 				int pc_reset, u_int8_t *param_buf,
 				u_int32_t param_len, u_int8_t sense_len,
 				u_int32_t timeout);
 
 void		scsi_prevent(struct ccb_scsiio *csio, u_int32_t retries,
 			     void (*cbfcnp)(struct cam_periph *, union ccb *),
 			     u_int8_t tag_action, u_int8_t action,
 			     u_int8_t sense_len, u_int32_t timeout);
 
 void		scsi_read_capacity(struct ccb_scsiio *csio, u_int32_t retries,
 				   void (*cbfcnp)(struct cam_periph *, 
 				   union ccb *), u_int8_t tag_action, 
 				   struct scsi_read_capacity_data *,
 				   u_int8_t sense_len, u_int32_t timeout);
 void		scsi_read_capacity_16(struct ccb_scsiio *csio, uint32_t retries,
 				      void (*cbfcnp)(struct cam_periph *,
 				      union ccb *), uint8_t tag_action,
 				      uint64_t lba, int reladr, int pmi,
 				      uint8_t *rcap_buf, int rcap_buf_len,
 				      uint8_t sense_len, uint32_t timeout);
 
 void		scsi_report_luns(struct ccb_scsiio *csio, u_int32_t retries,
 				 void (*cbfcnp)(struct cam_periph *, 
 				 union ccb *), u_int8_t tag_action, 
 				 u_int8_t select_report,
 				 struct scsi_report_luns_data *rpl_buf,
 				 u_int32_t alloc_len, u_int8_t sense_len,
 				 u_int32_t timeout);
 
 void		scsi_report_target_group(struct ccb_scsiio *csio, u_int32_t retries,
 				 void (*cbfcnp)(struct cam_periph *, 
 				 union ccb *), u_int8_t tag_action, 
 				 u_int8_t pdf,
 				 void *buf,
 				 u_int32_t alloc_len, u_int8_t sense_len,
 				 u_int32_t timeout);
 
 void		scsi_set_target_group(struct ccb_scsiio *csio, u_int32_t retries,
 				 void (*cbfcnp)(struct cam_periph *, 
 				 union ccb *), u_int8_t tag_action, void *buf,
 				 u_int32_t alloc_len, u_int8_t sense_len,
 				 u_int32_t timeout);
 
 void		scsi_synchronize_cache(struct ccb_scsiio *csio, 
 				       u_int32_t retries,
 				       void (*cbfcnp)(struct cam_periph *, 
 				       union ccb *), u_int8_t tag_action, 
 				       u_int32_t begin_lba, u_int16_t lb_count,
 				       u_int8_t sense_len, u_int32_t timeout);
 
 void scsi_receive_diagnostic_results(struct ccb_scsiio *csio, u_int32_t retries,
 				     void (*cbfcnp)(struct cam_periph *,
 						    union ccb*),
 				     uint8_t tag_action, int pcv,
 				     uint8_t page_code, uint8_t *data_ptr,
 				     uint16_t allocation_length,
 				     uint8_t sense_len, uint32_t timeout);
 
 void scsi_send_diagnostic(struct ccb_scsiio *csio, u_int32_t retries,
 			  void (*cbfcnp)(struct cam_periph *, union ccb *),
 			  uint8_t tag_action, int unit_offline,
 			  int device_offline, int self_test, int page_format,
 			  int self_test_code, uint8_t *data_ptr,
 			  uint16_t param_list_length, uint8_t sense_len,
 			  uint32_t timeout);
 
 void scsi_read_buffer(struct ccb_scsiio *csio, u_int32_t retries,
 			void (*cbfcnp)(struct cam_periph *, union ccb*),
 			uint8_t tag_action, int mode,
 			uint8_t buffer_id, u_int32_t offset,
 			uint8_t *data_ptr, uint32_t allocation_length,
 			uint8_t sense_len, uint32_t timeout);
 
 void scsi_write_buffer(struct ccb_scsiio *csio, u_int32_t retries,
 			void (*cbfcnp)(struct cam_periph *, union ccb *),
 			uint8_t tag_action, int mode,
 			uint8_t buffer_id, u_int32_t offset,
 			uint8_t *data_ptr, uint32_t param_list_length,
 			uint8_t sense_len, uint32_t timeout);
 
 #define	SCSI_RW_READ	0x0001
 #define	SCSI_RW_WRITE	0x0002
 #define	SCSI_RW_DIRMASK	0x0003
 #define	SCSI_RW_BIO	0x1000
 void scsi_read_write(struct ccb_scsiio *csio, u_int32_t retries,
 		     void (*cbfcnp)(struct cam_periph *, union ccb *),
 		     u_int8_t tag_action, int readop, u_int8_t byte2, 
 		     int minimum_cmd_size, u_int64_t lba,
 		     u_int32_t block_count, u_int8_t *data_ptr,
 		     u_int32_t dxfer_len, u_int8_t sense_len,
 		     u_int32_t timeout);
 
 void scsi_write_same(struct ccb_scsiio *csio, u_int32_t retries,
 		     void (*cbfcnp)(struct cam_periph *, union ccb *),
 		     u_int8_t tag_action, u_int8_t byte2, 
 		     int minimum_cmd_size, u_int64_t lba,
 		     u_int32_t block_count, u_int8_t *data_ptr,
 		     u_int32_t dxfer_len, u_int8_t sense_len,
 		     u_int32_t timeout);
 
 void scsi_ata_identify(struct ccb_scsiio *csio, u_int32_t retries,
 		       void (*cbfcnp)(struct cam_periph *, union ccb *),
 		       u_int8_t tag_action, u_int8_t *data_ptr,
 		       u_int16_t dxfer_len, u_int8_t sense_len,
 		       u_int32_t timeout);
 
 void scsi_ata_trim(struct ccb_scsiio *csio, u_int32_t retries,
 	           void (*cbfcnp)(struct cam_periph *, union ccb *),
 	           u_int8_t tag_action, u_int16_t block_count,
 	           u_int8_t *data_ptr, u_int16_t dxfer_len,
 	           u_int8_t sense_len, u_int32_t timeout);
 
 void scsi_ata_pass_16(struct ccb_scsiio *csio, u_int32_t retries,
 		      void (*cbfcnp)(struct cam_periph *, union ccb *),
 		      u_int32_t flags, u_int8_t tag_action,
 		      u_int8_t protocol, u_int8_t ata_flags, u_int16_t features,
 		      u_int16_t sector_count, uint64_t lba, u_int8_t command,
 		      u_int8_t control, u_int8_t *data_ptr, u_int16_t dxfer_len,
 		      u_int8_t sense_len, u_int32_t timeout);
 
 void scsi_unmap(struct ccb_scsiio *csio, u_int32_t retries,
 		void (*cbfcnp)(struct cam_periph *, union ccb *),
 		u_int8_t tag_action, u_int8_t byte2,
 		u_int8_t *data_ptr, u_int16_t dxfer_len,
 		u_int8_t sense_len, u_int32_t timeout);
 
 void scsi_start_stop(struct ccb_scsiio *csio, u_int32_t retries,
 		     void (*cbfcnp)(struct cam_periph *, union ccb *),
 		     u_int8_t tag_action, int start, int load_eject,
 		     int immediate, u_int8_t sense_len, u_int32_t timeout);
 
 void scsi_persistent_reserve_in(struct ccb_scsiio *csio, uint32_t retries, 
 				void (*cbfcnp)(struct cam_periph *,union ccb *),
 				uint8_t tag_action, int service_action,
 				uint8_t *data_ptr, uint32_t dxfer_len,
 				int sense_len, int timeout);
 
 void scsi_persistent_reserve_out(struct ccb_scsiio *csio, uint32_t retries, 
 				 void (*cbfcnp)(struct cam_periph *,
 				       union ccb *),
 				 uint8_t tag_action, int service_action,
 				 int scope, int res_type, uint8_t *data_ptr,
 				 uint32_t dxfer_len, int sense_len,
 				 int timeout);
 
 int		scsi_inquiry_match(caddr_t inqbuffer, caddr_t table_entry);
 int		scsi_static_inquiry_match(caddr_t inqbuffer,
 					  caddr_t table_entry);
 int		scsi_devid_match(uint8_t *rhs, size_t rhs_len,
 				 uint8_t *lhs, size_t lhs_len);
 
 void scsi_extract_sense(struct scsi_sense_data *sense, int *error_code,
 			int *sense_key, int *asc, int *ascq);
 int scsi_extract_sense_ccb(union ccb *ccb, int *error_code, int *sense_key,
 			   int *asc, int *ascq);
 void scsi_extract_sense_len(struct scsi_sense_data *sense,
 			    u_int sense_len, int *error_code, int *sense_key,
 			    int *asc, int *ascq, int show_errors);
 int scsi_get_sense_key(struct scsi_sense_data *sense, u_int sense_len,
 		       int show_errors);
 int scsi_get_asc(struct scsi_sense_data *sense, u_int sense_len,
 		 int show_errors);
 int scsi_get_ascq(struct scsi_sense_data *sense, u_int sense_len,
 		  int show_errors);
 static __inline void scsi_ulto2b(u_int32_t val, u_int8_t *bytes);
 static __inline void scsi_ulto3b(u_int32_t val, u_int8_t *bytes);
 static __inline void scsi_ulto4b(u_int32_t val, u_int8_t *bytes);
 static __inline void scsi_u64to8b(u_int64_t val, u_int8_t *bytes);
 static __inline uint32_t scsi_2btoul(const uint8_t *bytes);
 static __inline uint32_t scsi_3btoul(const uint8_t *bytes);
 static __inline int32_t scsi_3btol(const uint8_t *bytes);
 static __inline uint32_t scsi_4btoul(const uint8_t *bytes);
 static __inline uint64_t scsi_8btou64(const uint8_t *bytes);
 static __inline void *find_mode_page_6(struct scsi_mode_header_6 *mode_header);
 static __inline void *find_mode_page_10(struct scsi_mode_header_10 *mode_header);
 
 static __inline void
 scsi_ulto2b(u_int32_t val, u_int8_t *bytes)
 {
 
 	bytes[0] = (val >> 8) & 0xff;
 	bytes[1] = val & 0xff;
 }
 
 static __inline void
 scsi_ulto3b(u_int32_t val, u_int8_t *bytes)
 {
 
 	bytes[0] = (val >> 16) & 0xff;
 	bytes[1] = (val >> 8) & 0xff;
 	bytes[2] = val & 0xff;
 }
 
 static __inline void
 scsi_ulto4b(u_int32_t val, u_int8_t *bytes)
 {
 
 	bytes[0] = (val >> 24) & 0xff;
 	bytes[1] = (val >> 16) & 0xff;
 	bytes[2] = (val >> 8) & 0xff;
 	bytes[3] = val & 0xff;
 }
 
 static __inline void
 scsi_u64to8b(u_int64_t val, u_int8_t *bytes)
 {
 
 	bytes[0] = (val >> 56) & 0xff;
 	bytes[1] = (val >> 48) & 0xff;
 	bytes[2] = (val >> 40) & 0xff;
 	bytes[3] = (val >> 32) & 0xff;
 	bytes[4] = (val >> 24) & 0xff;
 	bytes[5] = (val >> 16) & 0xff;
 	bytes[6] = (val >> 8) & 0xff;
 	bytes[7] = val & 0xff;
 }
 
 static __inline uint32_t
 scsi_2btoul(const uint8_t *bytes)
 {
 	uint32_t rv;
 
 	rv = (bytes[0] << 8) |
 	     bytes[1];
 	return (rv);
 }
 
 static __inline uint32_t
 scsi_3btoul(const uint8_t *bytes)
 {
 	uint32_t rv;
 
 	rv = (bytes[0] << 16) |
 	     (bytes[1] << 8) |
 	     bytes[2];
 	return (rv);
 }
 
 static __inline int32_t 
 scsi_3btol(const uint8_t *bytes)
 {
 	uint32_t rc = scsi_3btoul(bytes);
  
 	if (rc & 0x00800000)
 		rc |= 0xff000000;
 
 	return (int32_t) rc;
 }
 
 static __inline uint32_t
 scsi_4btoul(const uint8_t *bytes)
 {
 	uint32_t rv;
 
 	rv = (bytes[0] << 24) |
 	     (bytes[1] << 16) |
 	     (bytes[2] << 8) |
 	     bytes[3];
 	return (rv);
 }
 
 static __inline uint64_t
 scsi_8btou64(const uint8_t *bytes)
 {
         uint64_t rv;
  
 	rv = (((uint64_t)bytes[0]) << 56) |
 	     (((uint64_t)bytes[1]) << 48) |
 	     (((uint64_t)bytes[2]) << 40) |
 	     (((uint64_t)bytes[3]) << 32) |
 	     (((uint64_t)bytes[4]) << 24) |
 	     (((uint64_t)bytes[5]) << 16) |
 	     (((uint64_t)bytes[6]) << 8) |
 	     bytes[7];
 	return (rv);
 }
 
 /*
  * Given the pointer to a returned mode sense buffer, return a pointer to
  * the start of the first mode page.
  */
 static __inline void *
 find_mode_page_6(struct scsi_mode_header_6 *mode_header)
 {
 	void *page_start;
 
 	page_start = (void *)((u_int8_t *)&mode_header[1] +
 			      mode_header->blk_desc_len);
 
 	return(page_start);
 }
 
 static __inline void *
 find_mode_page_10(struct scsi_mode_header_10 *mode_header)
 {
 	void *page_start;
 
 	page_start = (void *)((u_int8_t *)&mode_header[1] +
 			       scsi_2btoul(mode_header->blk_desc_len));
 
 	return(page_start);
 }
 
 __END_DECLS
 
 #endif /*_SCSI_SCSI_ALL_H*/
Index: projects/clang360-import/sys/cddl/contrib/opensolaris/uts/common/dtrace/fasttrap.c
===================================================================
--- projects/clang360-import/sys/cddl/contrib/opensolaris/uts/common/dtrace/fasttrap.c	(revision 277944)
+++ projects/clang360-import/sys/cddl/contrib/opensolaris/uts/common/dtrace/fasttrap.c	(revision 277945)
@@ -1,2728 +1,2733 @@
 /*
  * CDDL HEADER START
  *
  * The contents of this file are subject to the terms of the
  * Common Development and Distribution License (the "License").
  * You may not use this file except in compliance with the License.
  *
  * You can obtain a copy of the license at usr/src/OPENSOLARIS.LICENSE
  * or http://www.opensolaris.org/os/licensing.
  * See the License for the specific language governing permissions
  * and limitations under the License.
  *
  * When distributing Covered Code, include this CDDL HEADER in each
  * file and include the License file at usr/src/OPENSOLARIS.LICENSE.
  * If applicable, add the following below this CDDL HEADER, with the
  * fields enclosed by brackets "[]" replaced with your own identifying
  * information: Portions Copyright [yyyy] [name of copyright owner]
  *
  * CDDL HEADER END
  *
  * Portions Copyright 2010 The FreeBSD Foundation
  *
  * $FreeBSD$
  */
 
 /*
  * Copyright 2008 Sun Microsystems, Inc.  All rights reserved.
  * Use is subject to license terms.
  */
 
 /*
  * Copyright (c) 2013, Joyent, Inc. All rights reserved.
  */
 
 #include <sys/atomic.h>
 #include <sys/errno.h>
 #include <sys/stat.h>
 #include <sys/modctl.h>
 #include <sys/conf.h>
 #include <sys/systm.h>
 #ifdef illumos
 #include <sys/ddi.h>
 #endif
 #include <sys/sunddi.h>
 #include <sys/cpuvar.h>
 #include <sys/kmem.h>
 #ifdef illumos
 #include <sys/strsubr.h>
 #endif
 #include <sys/fasttrap.h>
 #include <sys/fasttrap_impl.h>
 #include <sys/fasttrap_isa.h>
 #include <sys/dtrace.h>
 #include <sys/dtrace_impl.h>
 #include <sys/sysmacros.h>
 #include <sys/proc.h>
 #include <sys/policy.h>
 #ifdef illumos
 #include <util/qsort.h>
 #endif
 #include <sys/mutex.h>
 #include <sys/kernel.h>
 #ifndef illumos
 #include <sys/dtrace_bsd.h>
 #include <sys/eventhandler.h>
 #include <sys/u8_textprep.h>
 #include <sys/user.h>
 #include <vm/vm.h>
 #include <vm/pmap.h>
 #include <vm/vm_map.h>
 #include <vm/vm_param.h>
 #include <cddl/dev/dtrace/dtrace_cddl.h>
 #endif
 
 /*
  * User-Land Trap-Based Tracing
  * ----------------------------
  *
  * The fasttrap provider allows DTrace consumers to instrument any user-level
  * instruction to gather data; this includes probes with semantic
  * signifigance like entry and return as well as simple offsets into the
  * function. While the specific techniques used are very ISA specific, the
  * methodology is generalizable to any architecture.
  *
  *
  * The General Methodology
  * -----------------------
  *
  * With the primary goal of tracing every user-land instruction and the
  * limitation that we can't trust user space so don't want to rely on much
  * information there, we begin by replacing the instructions we want to trace
  * with trap instructions. Each instruction we overwrite is saved into a hash
  * table keyed by process ID and pc address. When we enter the kernel due to
  * this trap instruction, we need the effects of the replaced instruction to
  * appear to have occurred before we proceed with the user thread's
  * execution.
  *
  * Each user level thread is represented by a ulwp_t structure which is
  * always easily accessible through a register. The most basic way to produce
  * the effects of the instruction we replaced is to copy that instruction out
  * to a bit of scratch space reserved in the user thread's ulwp_t structure
  * (a sort of kernel-private thread local storage), set the PC to that
  * scratch space and single step. When we reenter the kernel after single
  * stepping the instruction we must then adjust the PC to point to what would
  * normally be the next instruction. Of course, special care must be taken
  * for branches and jumps, but these represent such a small fraction of any
  * instruction set that writing the code to emulate these in the kernel is
  * not too difficult.
  *
  * Return probes may require several tracepoints to trace every return site,
  * and, conversely, each tracepoint may activate several probes (the entry
  * and offset 0 probes, for example). To solve this muliplexing problem,
  * tracepoints contain lists of probes to activate and probes contain lists
  * of tracepoints to enable. If a probe is activated, it adds its ID to
  * existing tracepoints or creates new ones as necessary.
  *
  * Most probes are activated _before_ the instruction is executed, but return
  * probes are activated _after_ the effects of the last instruction of the
  * function are visible. Return probes must be fired _after_ we have
  * single-stepped the instruction whereas all other probes are fired
  * beforehand.
  *
  *
  * Lock Ordering
  * -------------
  *
  * The lock ordering below -- both internally and with respect to the DTrace
  * framework -- is a little tricky and bears some explanation. Each provider
  * has a lock (ftp_mtx) that protects its members including reference counts
  * for enabled probes (ftp_rcount), consumers actively creating probes
  * (ftp_ccount) and USDT consumers (ftp_mcount); all three prevent a provider
  * from being freed. A provider is looked up by taking the bucket lock for the
  * provider hash table, and is returned with its lock held. The provider lock
  * may be taken in functions invoked by the DTrace framework, but may not be
  * held while calling functions in the DTrace framework.
  *
  * To ensure consistency over multiple calls to the DTrace framework, the
  * creation lock (ftp_cmtx) should be held. Naturally, the creation lock may
  * not be taken when holding the provider lock as that would create a cyclic
  * lock ordering. In situations where one would naturally take the provider
  * lock and then the creation lock, we instead up a reference count to prevent
  * the provider from disappearing, drop the provider lock, and acquire the
  * creation lock.
  *
  * Briefly:
  * 	bucket lock before provider lock
  *	DTrace before provider lock
  *	creation lock before DTrace
  *	never hold the provider lock and creation lock simultaneously
  */
 
 static d_open_t fasttrap_open;
 static d_ioctl_t fasttrap_ioctl;
 
 static struct cdevsw fasttrap_cdevsw = {
 	.d_version	= D_VERSION,
 	.d_open		= fasttrap_open,
 	.d_ioctl	= fasttrap_ioctl,
 	.d_name		= "fasttrap",
 };
 static struct cdev *fasttrap_cdev;
 static dtrace_meta_provider_id_t fasttrap_meta_id;
 
 static struct proc *fasttrap_cleanup_proc;
 static struct mtx fasttrap_cleanup_mtx;
 static uint_t fasttrap_cleanup_work, fasttrap_cleanup_drain, fasttrap_cleanup_cv;
 
 /*
  * Generation count on modifications to the global tracepoint lookup table.
  */
 static volatile uint64_t fasttrap_mod_gen;
 
 /*
  * When the fasttrap provider is loaded, fasttrap_max is set to either
  * FASTTRAP_MAX_DEFAULT or the value for fasttrap-max-probes in the
  * fasttrap.conf file. Each time a probe is created, fasttrap_total is
  * incremented by the number of tracepoints that may be associated with that
  * probe; fasttrap_total is capped at fasttrap_max.
  */
 #define	FASTTRAP_MAX_DEFAULT		250000
 static uint32_t fasttrap_max;
 static uint32_t fasttrap_total;
 
 /*
  * Copyright (c) 2011, Joyent, Inc. All rights reserved.
  */
 
 #define	FASTTRAP_TPOINTS_DEFAULT_SIZE	0x4000
 #define	FASTTRAP_PROVIDERS_DEFAULT_SIZE	0x100
 #define	FASTTRAP_PROCS_DEFAULT_SIZE	0x100
 
 #define	FASTTRAP_PID_NAME		"pid"
 
 fasttrap_hash_t			fasttrap_tpoints;
 static fasttrap_hash_t		fasttrap_provs;
 static fasttrap_hash_t		fasttrap_procs;
 
 static uint64_t			fasttrap_pid_count;	/* pid ref count */
 static kmutex_t			fasttrap_count_mtx;	/* lock on ref count */
 
 #define	FASTTRAP_ENABLE_FAIL	1
 #define	FASTTRAP_ENABLE_PARTIAL	2
 
 static int fasttrap_tracepoint_enable(proc_t *, fasttrap_probe_t *, uint_t);
 static void fasttrap_tracepoint_disable(proc_t *, fasttrap_probe_t *, uint_t);
 
 static fasttrap_provider_t *fasttrap_provider_lookup(pid_t, const char *,
     const dtrace_pattr_t *);
 static void fasttrap_provider_retire(pid_t, const char *, int);
 static void fasttrap_provider_free(fasttrap_provider_t *);
 
 static fasttrap_proc_t *fasttrap_proc_lookup(pid_t);
 static void fasttrap_proc_release(fasttrap_proc_t *);
 
 #ifndef illumos
 static void fasttrap_thread_dtor(void *, struct thread *);
 #endif
 
 #define	FASTTRAP_PROVS_INDEX(pid, name) \
 	((fasttrap_hash_str(name) + (pid)) & fasttrap_provs.fth_mask)
 
 #define	FASTTRAP_PROCS_INDEX(pid) ((pid) & fasttrap_procs.fth_mask)
 
 #ifndef illumos
 static kmutex_t fasttrap_cpuc_pid_lock[MAXCPU];
 static eventhandler_tag fasttrap_thread_dtor_tag;
 #endif
 
 static int
 fasttrap_highbit(ulong_t i)
 {
 	int h = 1;
 
 	if (i == 0)
 		return (0);
 #ifdef _LP64
 	if (i & 0xffffffff00000000ul) {
 		h += 32; i >>= 32;
 	}
 #endif
 	if (i & 0xffff0000) {
 		h += 16; i >>= 16;
 	}
 	if (i & 0xff00) {
 		h += 8; i >>= 8;
 	}
 	if (i & 0xf0) {
 		h += 4; i >>= 4;
 	}
 	if (i & 0xc) {
 		h += 2; i >>= 2;
 	}
 	if (i & 0x2) {
 		h += 1;
 	}
 	return (h);
 }
 
 static uint_t
 fasttrap_hash_str(const char *p)
 {
 	unsigned int g;
 	uint_t hval = 0;
 
 	while (*p) {
 		hval = (hval << 4) + *p++;
 		if ((g = (hval & 0xf0000000)) != 0)
 			hval ^= g >> 24;
 		hval &= ~g;
 	}
 	return (hval);
 }
 
 void
 fasttrap_sigtrap(proc_t *p, kthread_t *t, uintptr_t pc)
 {
 #ifdef illumos
 	sigqueue_t *sqp = kmem_zalloc(sizeof (sigqueue_t), KM_SLEEP);
 
 	sqp->sq_info.si_signo = SIGTRAP;
 	sqp->sq_info.si_code = TRAP_DTRACE;
 	sqp->sq_info.si_addr = (caddr_t)pc;
 
 	mutex_enter(&p->p_lock);
 	sigaddqa(p, t, sqp);
 	mutex_exit(&p->p_lock);
 
 	if (t != NULL)
 		aston(t);
 #else
 	ksiginfo_t *ksi = kmem_zalloc(sizeof (ksiginfo_t), KM_SLEEP);
 
 	ksiginfo_init(ksi);
 	ksi->ksi_signo = SIGTRAP;
 	ksi->ksi_code = TRAP_DTRACE;
 	ksi->ksi_addr = (caddr_t)pc;
 	PROC_LOCK(p);
-	(void) tdksignal(t, SIGTRAP, ksi);
+	(void) tdsendsignal(p, t, SIGTRAP, ksi);
 	PROC_UNLOCK(p);
 #endif
 }
 
 #ifndef illumos
 /*
  * Obtain a chunk of scratch space in the address space of the target process.
  */
 fasttrap_scrspace_t *
 fasttrap_scraddr(struct thread *td, fasttrap_proc_t *fprc)
 {
 	fasttrap_scrblock_t *scrblk;
 	fasttrap_scrspace_t *scrspc;
 	struct proc *p;
 	vm_offset_t addr;
 	int error, i;
 
 	scrspc = NULL;
 	if (td->t_dtrace_sscr != NULL) {
 		/* If the thread already has scratch space, we're done. */
 		scrspc = (fasttrap_scrspace_t *)td->t_dtrace_sscr;
 		return (scrspc);
 	}
 
 	p = td->td_proc;
 
 	mutex_enter(&fprc->ftpc_mtx);
 	if (LIST_EMPTY(&fprc->ftpc_fscr)) {
 		/*
 		 * No scratch space is available, so we'll map a new scratch
 		 * space block into the traced process' address space.
 		 */
 		addr = 0;
 		error = vm_map_find(&p->p_vmspace->vm_map, NULL, 0, &addr,
 		    FASTTRAP_SCRBLOCK_SIZE, 0, VMFS_ANY_SPACE, VM_PROT_ALL,
 		    VM_PROT_ALL, 0);
 		if (error != KERN_SUCCESS)
 			goto done;
 
 		scrblk = malloc(sizeof(*scrblk), M_SOLARIS, M_WAITOK);
 		scrblk->ftsb_addr = addr;
 		LIST_INSERT_HEAD(&fprc->ftpc_scrblks, scrblk, ftsb_next);
 
 		/*
 		 * Carve the block up into chunks and put them on the free list.
 		 */
 		for (i = 0;
 		    i < FASTTRAP_SCRBLOCK_SIZE / FASTTRAP_SCRSPACE_SIZE; i++) {
 			scrspc = malloc(sizeof(*scrspc), M_SOLARIS, M_WAITOK);
 			scrspc->ftss_addr = addr +
 			    i * FASTTRAP_SCRSPACE_SIZE;
 			LIST_INSERT_HEAD(&fprc->ftpc_fscr, scrspc,
 			    ftss_next);
 		}
 	}
 
 	/*
 	 * Take the first scratch chunk off the free list, put it on the
 	 * allocated list, and return its address.
 	 */
 	scrspc = LIST_FIRST(&fprc->ftpc_fscr);
 	LIST_REMOVE(scrspc, ftss_next);
 	LIST_INSERT_HEAD(&fprc->ftpc_ascr, scrspc, ftss_next);
 
 	/*
 	 * This scratch space is reserved for use by td until the thread exits.
 	 */
 	td->t_dtrace_sscr = scrspc;
 
 done:
 	mutex_exit(&fprc->ftpc_mtx);
 
 	return (scrspc);
 }
 
 /*
  * Return any allocated per-thread scratch space chunks back to the process'
  * free list.
  */
 static void
 fasttrap_thread_dtor(void *arg __unused, struct thread *td)
 {
 	fasttrap_bucket_t *bucket;
 	fasttrap_proc_t *fprc;
 	fasttrap_scrspace_t *scrspc;
 	pid_t pid;
 
 	if (td->t_dtrace_sscr == NULL)
 		return;
 
 	pid = td->td_proc->p_pid;
 	bucket = &fasttrap_procs.fth_table[FASTTRAP_PROCS_INDEX(pid)];
 	fprc = NULL;
 
 	/* Look up the fasttrap process handle for this process. */
 	mutex_enter(&bucket->ftb_mtx);
 	for (fprc = bucket->ftb_data; fprc != NULL; fprc = fprc->ftpc_next) {
 		if (fprc->ftpc_pid == pid) {
 			mutex_enter(&fprc->ftpc_mtx);
 			mutex_exit(&bucket->ftb_mtx);
 			break;
 		}
 	}
 	if (fprc == NULL) {
 		mutex_exit(&bucket->ftb_mtx);
 		return;
 	}
 
 	scrspc = (fasttrap_scrspace_t *)td->t_dtrace_sscr;
 	LIST_REMOVE(scrspc, ftss_next);
 	LIST_INSERT_HEAD(&fprc->ftpc_fscr, scrspc, ftss_next);
 
 	mutex_exit(&fprc->ftpc_mtx);
 }
 #endif
 
 /*
  * This function ensures that no threads are actively using the memory
  * associated with probes that were formerly live.
  */
 static void
 fasttrap_mod_barrier(uint64_t gen)
 {
 	int i;
 
 	if (gen < fasttrap_mod_gen)
 		return;
 
 	fasttrap_mod_gen++;
 
 	CPU_FOREACH(i) {
 		mutex_enter(&fasttrap_cpuc_pid_lock[i]);
 		mutex_exit(&fasttrap_cpuc_pid_lock[i]);
 	}
 }
 
 /*
  * This function performs asynchronous cleanup of fasttrap providers. The
  * Solaris implementation of this mechanism use a timeout that's activated in
  * fasttrap_pid_cleanup(), but this doesn't work in FreeBSD: one may sleep while
  * holding the DTrace mutexes, but it is unsafe to sleep in a callout handler.
  * Thus we use a dedicated process to perform the cleanup when requested.
  */
 /*ARGSUSED*/
 static void
 fasttrap_pid_cleanup_cb(void *data)
 {
 	fasttrap_provider_t **fpp, *fp;
 	fasttrap_bucket_t *bucket;
 	dtrace_provider_id_t provid;
 	int i, later = 0, rval;
 
 	mtx_lock(&fasttrap_cleanup_mtx);
 	while (!fasttrap_cleanup_drain || later > 0) {
 		fasttrap_cleanup_work = 0;
 		mtx_unlock(&fasttrap_cleanup_mtx);
 
 		later = 0;
 
 		/*
 		 * Iterate over all the providers trying to remove the marked
 		 * ones. If a provider is marked but not retired, we just
 		 * have to take a crack at removing it -- it's no big deal if
 		 * we can't.
 		 */
 		for (i = 0; i < fasttrap_provs.fth_nent; i++) {
 			bucket = &fasttrap_provs.fth_table[i];
 			mutex_enter(&bucket->ftb_mtx);
 			fpp = (fasttrap_provider_t **)&bucket->ftb_data;
 
 			while ((fp = *fpp) != NULL) {
 				if (!fp->ftp_marked) {
 					fpp = &fp->ftp_next;
 					continue;
 				}
 
 				mutex_enter(&fp->ftp_mtx);
 
 				/*
 				 * If this provider has consumers actively
 				 * creating probes (ftp_ccount) or is a USDT
 				 * provider (ftp_mcount), we can't unregister
 				 * or even condense.
 				 */
 				if (fp->ftp_ccount != 0 ||
 				    fp->ftp_mcount != 0) {
 					mutex_exit(&fp->ftp_mtx);
 					fp->ftp_marked = 0;
 					continue;
 				}
 
 				if (!fp->ftp_retired || fp->ftp_rcount != 0)
 					fp->ftp_marked = 0;
 
 				mutex_exit(&fp->ftp_mtx);
 
 				/*
 				 * If we successfully unregister this
 				 * provider we can remove it from the hash
 				 * chain and free the memory. If our attempt
 				 * to unregister fails and this is a retired
 				 * provider, increment our flag to try again
 				 * pretty soon. If we've consumed more than
 				 * half of our total permitted number of
 				 * probes call dtrace_condense() to try to
 				 * clean out the unenabled probes.
 				 */
 				provid = fp->ftp_provid;
 				if ((rval = dtrace_unregister(provid)) != 0) {
 					if (fasttrap_total > fasttrap_max / 2)
 						(void) dtrace_condense(provid);
 
 					if (rval == EAGAIN)
 						fp->ftp_marked = 1;
 
 					later += fp->ftp_marked;
 					fpp = &fp->ftp_next;
 				} else {
 					*fpp = fp->ftp_next;
 					fasttrap_provider_free(fp);
 				}
 			}
 			mutex_exit(&bucket->ftb_mtx);
 		}
 		mtx_lock(&fasttrap_cleanup_mtx);
 
 		/*
 		 * If we were unable to retire a provider, try again after a
 		 * second. This situation can occur in certain circumstances
 		 * where providers cannot be unregistered even though they have
 		 * no probes enabled because of an execution of dtrace -l or
 		 * something similar.
 		 */
 		if (later > 0 || fasttrap_cleanup_work ||
 		    fasttrap_cleanup_drain) {
 			mtx_unlock(&fasttrap_cleanup_mtx);
 			pause("ftclean", hz);
 			mtx_lock(&fasttrap_cleanup_mtx);
 		} else
 			mtx_sleep(&fasttrap_cleanup_cv, &fasttrap_cleanup_mtx,
 			    0, "ftcl", 0);
 	}
 
 	/*
 	 * Wake up the thread in fasttrap_unload() now that we're done.
 	 */
 	wakeup(&fasttrap_cleanup_drain);
 	mtx_unlock(&fasttrap_cleanup_mtx);
 
 	kthread_exit();
 }
 
 /*
  * Activates the asynchronous cleanup mechanism.
  */
 static void
 fasttrap_pid_cleanup(void)
 {
 
 	mtx_lock(&fasttrap_cleanup_mtx);
 	if (!fasttrap_cleanup_work) {
 		fasttrap_cleanup_work = 1;
 		wakeup(&fasttrap_cleanup_cv);
 	}
 	mtx_unlock(&fasttrap_cleanup_mtx);
 }
 
 /*
  * This is called from cfork() via dtrace_fasttrap_fork(). The child
  * process's address space is (roughly) a copy of the parent process's so
  * we have to remove all the instrumentation we had previously enabled in the
  * parent.
  */
 static void
 fasttrap_fork(proc_t *p, proc_t *cp)
 {
 #ifndef illumos
 	fasttrap_scrblock_t *scrblk;
 	fasttrap_proc_t *fprc = NULL;
 #endif
 	pid_t ppid = p->p_pid;
 	int i;
 
 #ifdef illumos
 	ASSERT(curproc == p);
 	ASSERT(p->p_proc_flag & P_PR_LOCK);
 #else
 	PROC_LOCK_ASSERT(p, MA_OWNED);
 #endif
 #ifdef illumos
 	ASSERT(p->p_dtrace_count > 0);
 #else
 	if (p->p_dtrace_helpers) {
 		/*
 		 * dtrace_helpers_duplicate() allocates memory.
 		 */
 		_PHOLD(cp);
 		PROC_UNLOCK(p);
 		PROC_UNLOCK(cp);
 		dtrace_helpers_duplicate(p, cp);
 		PROC_LOCK(cp);
 		PROC_LOCK(p);
 		_PRELE(cp);
 	}
 	/*
 	 * This check is purposely here instead of in kern_fork.c because,
 	 * for legal resons, we cannot include the dtrace_cddl.h header
 	 * inside kern_fork.c and insert if-clause there.
 	 */
 	if (p->p_dtrace_count == 0)
 		return;
 #endif
 	ASSERT(cp->p_dtrace_count == 0);
 
 	/*
 	 * This would be simpler and faster if we maintained per-process
 	 * hash tables of enabled tracepoints. It could, however, potentially
 	 * slow down execution of a tracepoint since we'd need to go
 	 * through two levels of indirection. In the future, we should
 	 * consider either maintaining per-process ancillary lists of
 	 * enabled tracepoints or hanging a pointer to a per-process hash
 	 * table of enabled tracepoints off the proc structure.
 	 */
 
 	/*
 	 * We don't have to worry about the child process disappearing
 	 * because we're in fork().
 	 */
 #ifdef illumos
 	mtx_lock_spin(&cp->p_slock);
 	sprlock_proc(cp);
 	mtx_unlock_spin(&cp->p_slock);
 #else
 	/*
 	 * fasttrap_tracepoint_remove() expects the child process to be
 	 * unlocked and the VM then expects curproc to be unlocked.
 	 */
 	_PHOLD(cp);
 	PROC_UNLOCK(cp);
 	PROC_UNLOCK(p);
 #endif
 
 	/*
 	 * Iterate over every tracepoint looking for ones that belong to the
 	 * parent process, and remove each from the child process.
 	 */
 	for (i = 0; i < fasttrap_tpoints.fth_nent; i++) {
 		fasttrap_tracepoint_t *tp;
 		fasttrap_bucket_t *bucket = &fasttrap_tpoints.fth_table[i];
 
 		mutex_enter(&bucket->ftb_mtx);
 		for (tp = bucket->ftb_data; tp != NULL; tp = tp->ftt_next) {
 			if (tp->ftt_pid == ppid &&
 			    tp->ftt_proc->ftpc_acount != 0) {
 				int ret = fasttrap_tracepoint_remove(cp, tp);
 				ASSERT(ret == 0);
 
 				/*
 				 * The count of active providers can only be
 				 * decremented (i.e. to zero) during exec,
 				 * exit, and removal of a meta provider so it
 				 * should be impossible to drop the count
 				 * mid-fork.
 				 */
 				ASSERT(tp->ftt_proc->ftpc_acount != 0);
 #ifndef illumos
 				fprc = tp->ftt_proc;
 #endif
 			}
 		}
 		mutex_exit(&bucket->ftb_mtx);
 
 #ifndef illumos
 		/*
 		 * Unmap any scratch space inherited from the parent's address
 		 * space.
 		 */
 		if (fprc != NULL) {
 			mutex_enter(&fprc->ftpc_mtx);
 			LIST_FOREACH(scrblk, &fprc->ftpc_scrblks, ftsb_next) {
 				vm_map_remove(&cp->p_vmspace->vm_map,
 				    scrblk->ftsb_addr,
 				    scrblk->ftsb_addr + FASTTRAP_SCRBLOCK_SIZE);
 			}
 			mutex_exit(&fprc->ftpc_mtx);
 		}
 #endif
 	}
 
 #ifdef illumos
 	mutex_enter(&cp->p_lock);
 	sprunlock(cp);
 #else
 	PROC_LOCK(p);
 	PROC_LOCK(cp);
 	_PRELE(cp);
 #endif
 }
 
 /*
  * This is called from proc_exit() or from exec_common() if p_dtrace_probes
  * is set on the proc structure to indicate that there is a pid provider
  * associated with this process.
  */
 static void
 fasttrap_exec_exit(proc_t *p)
 {
 #ifndef illumos
 	struct thread *td;
 #endif
 
 #ifdef illumos
 	ASSERT(p == curproc);
 #else
 	PROC_LOCK_ASSERT(p, MA_OWNED);
 	_PHOLD(p);
 	/*
 	 * Since struct threads may be recycled, we cannot rely on t_dtrace_sscr
 	 * fields to be zeroed by kdtrace_thread_ctor. Thus we must zero it
 	 * ourselves when a process exits.
 	 */
 	FOREACH_THREAD_IN_PROC(p, td)
 		td->t_dtrace_sscr = NULL;
 	PROC_UNLOCK(p);
 #endif
 
 	/*
 	 * We clean up the pid provider for this process here; user-land
 	 * static probes are handled by the meta-provider remove entry point.
 	 */
 	fasttrap_provider_retire(p->p_pid, FASTTRAP_PID_NAME, 0);
 #ifndef illumos
 	if (p->p_dtrace_helpers)
 		dtrace_helpers_destroy(p);
 	PROC_LOCK(p);
 	_PRELE(p);
 #endif
 }
 
 
 /*ARGSUSED*/
 static void
 fasttrap_pid_provide(void *arg, dtrace_probedesc_t *desc)
 {
 	/*
 	 * There are no "default" pid probes.
 	 */
 }
 
 static int
 fasttrap_tracepoint_enable(proc_t *p, fasttrap_probe_t *probe, uint_t index)
 {
 	fasttrap_tracepoint_t *tp, *new_tp = NULL;
 	fasttrap_bucket_t *bucket;
 	fasttrap_id_t *id;
 	pid_t pid;
 	uintptr_t pc;
 
 	ASSERT(index < probe->ftp_ntps);
 
 	pid = probe->ftp_pid;
 	pc = probe->ftp_tps[index].fit_tp->ftt_pc;
 	id = &probe->ftp_tps[index].fit_id;
 
 	ASSERT(probe->ftp_tps[index].fit_tp->ftt_pid == pid);
 
 #ifdef illumos
 	ASSERT(!(p->p_flag & SVFORK));
 #endif
 
 	/*
 	 * Before we make any modifications, make sure we've imposed a barrier
 	 * on the generation in which this probe was last modified.
 	 */
 	fasttrap_mod_barrier(probe->ftp_gen);
 
 	bucket = &fasttrap_tpoints.fth_table[FASTTRAP_TPOINTS_INDEX(pid, pc)];
 
 	/*
 	 * If the tracepoint has already been enabled, just add our id to the
 	 * list of interested probes. This may be our second time through
 	 * this path in which case we'll have constructed the tracepoint we'd
 	 * like to install. If we can't find a match, and have an allocated
 	 * tracepoint ready to go, enable that one now.
 	 *
 	 * A tracepoint whose process is defunct is also considered defunct.
 	 */
 again:
 	mutex_enter(&bucket->ftb_mtx);
 	for (tp = bucket->ftb_data; tp != NULL; tp = tp->ftt_next) {
 		/*
 		 * Note that it's safe to access the active count on the
 		 * associated proc structure because we know that at least one
 		 * provider (this one) will still be around throughout this
 		 * operation.
 		 */
 		if (tp->ftt_pid != pid || tp->ftt_pc != pc ||
 		    tp->ftt_proc->ftpc_acount == 0)
 			continue;
 
 		/*
 		 * Now that we've found a matching tracepoint, it would be
 		 * a decent idea to confirm that the tracepoint is still
 		 * enabled and the trap instruction hasn't been overwritten.
 		 * Since this is a little hairy, we'll punt for now.
 		 */
 
 		/*
 		 * This can't be the first interested probe. We don't have
 		 * to worry about another thread being in the midst of
 		 * deleting this tracepoint (which would be the only valid
 		 * reason for a tracepoint to have no interested probes)
 		 * since we're holding P_PR_LOCK for this process.
 		 */
 		ASSERT(tp->ftt_ids != NULL || tp->ftt_retids != NULL);
 
 		switch (id->fti_ptype) {
 		case DTFTP_ENTRY:
 		case DTFTP_OFFSETS:
 		case DTFTP_IS_ENABLED:
 			id->fti_next = tp->ftt_ids;
 			membar_producer();
 			tp->ftt_ids = id;
 			membar_producer();
 			break;
 
 		case DTFTP_RETURN:
 		case DTFTP_POST_OFFSETS:
 			id->fti_next = tp->ftt_retids;
 			membar_producer();
 			tp->ftt_retids = id;
 			membar_producer();
 			break;
 
 		default:
 			ASSERT(0);
 		}
 
 		mutex_exit(&bucket->ftb_mtx);
 
 		if (new_tp != NULL) {
 			new_tp->ftt_ids = NULL;
 			new_tp->ftt_retids = NULL;
 		}
 
 		return (0);
 	}
 
 	/*
 	 * If we have a good tracepoint ready to go, install it now while
 	 * we have the lock held and no one can screw with us.
 	 */
 	if (new_tp != NULL) {
 		int rc = 0;
 
 		new_tp->ftt_next = bucket->ftb_data;
 		membar_producer();
 		bucket->ftb_data = new_tp;
 		membar_producer();
 		mutex_exit(&bucket->ftb_mtx);
 
 		/*
 		 * Activate the tracepoint in the ISA-specific manner.
 		 * If this fails, we need to report the failure, but
 		 * indicate that this tracepoint must still be disabled
 		 * by calling fasttrap_tracepoint_disable().
 		 */
 		if (fasttrap_tracepoint_install(p, new_tp) != 0)
 			rc = FASTTRAP_ENABLE_PARTIAL;
 
 		/*
 		 * Increment the count of the number of tracepoints active in
 		 * the victim process.
 		 */
 #ifdef illumos
 		ASSERT(p->p_proc_flag & P_PR_LOCK);
 #endif
 		p->p_dtrace_count++;
 
 		return (rc);
 	}
 
 	mutex_exit(&bucket->ftb_mtx);
 
 	/*
 	 * Initialize the tracepoint that's been preallocated with the probe.
 	 */
 	new_tp = probe->ftp_tps[index].fit_tp;
 
 	ASSERT(new_tp->ftt_pid == pid);
 	ASSERT(new_tp->ftt_pc == pc);
 	ASSERT(new_tp->ftt_proc == probe->ftp_prov->ftp_proc);
 	ASSERT(new_tp->ftt_ids == NULL);
 	ASSERT(new_tp->ftt_retids == NULL);
 
 	switch (id->fti_ptype) {
 	case DTFTP_ENTRY:
 	case DTFTP_OFFSETS:
 	case DTFTP_IS_ENABLED:
 		id->fti_next = NULL;
 		new_tp->ftt_ids = id;
 		break;
 
 	case DTFTP_RETURN:
 	case DTFTP_POST_OFFSETS:
 		id->fti_next = NULL;
 		new_tp->ftt_retids = id;
 		break;
 
 	default:
 		ASSERT(0);
 	}
 
 	/*
 	 * If the ISA-dependent initialization goes to plan, go back to the
 	 * beginning and try to install this freshly made tracepoint.
 	 */
 	if (fasttrap_tracepoint_init(p, new_tp, pc, id->fti_ptype) == 0)
 		goto again;
 
 	new_tp->ftt_ids = NULL;
 	new_tp->ftt_retids = NULL;
 
 	return (FASTTRAP_ENABLE_FAIL);
 }
 
 static void
 fasttrap_tracepoint_disable(proc_t *p, fasttrap_probe_t *probe, uint_t index)
 {
 	fasttrap_bucket_t *bucket;
 	fasttrap_provider_t *provider = probe->ftp_prov;
 	fasttrap_tracepoint_t **pp, *tp;
 	fasttrap_id_t *id, **idp = NULL;
 	pid_t pid;
 	uintptr_t pc;
 
 	ASSERT(index < probe->ftp_ntps);
 
 	pid = probe->ftp_pid;
 	pc = probe->ftp_tps[index].fit_tp->ftt_pc;
 	id = &probe->ftp_tps[index].fit_id;
 
 	ASSERT(probe->ftp_tps[index].fit_tp->ftt_pid == pid);
 
 	/*
 	 * Find the tracepoint and make sure that our id is one of the
 	 * ones registered with it.
 	 */
 	bucket = &fasttrap_tpoints.fth_table[FASTTRAP_TPOINTS_INDEX(pid, pc)];
 	mutex_enter(&bucket->ftb_mtx);
 	for (tp = bucket->ftb_data; tp != NULL; tp = tp->ftt_next) {
 		if (tp->ftt_pid == pid && tp->ftt_pc == pc &&
 		    tp->ftt_proc == provider->ftp_proc)
 			break;
 	}
 
 	/*
 	 * If we somehow lost this tracepoint, we're in a world of hurt.
 	 */
 	ASSERT(tp != NULL);
 
 	switch (id->fti_ptype) {
 	case DTFTP_ENTRY:
 	case DTFTP_OFFSETS:
 	case DTFTP_IS_ENABLED:
 		ASSERT(tp->ftt_ids != NULL);
 		idp = &tp->ftt_ids;
 		break;
 
 	case DTFTP_RETURN:
 	case DTFTP_POST_OFFSETS:
 		ASSERT(tp->ftt_retids != NULL);
 		idp = &tp->ftt_retids;
 		break;
 
 	default:
 		ASSERT(0);
 	}
 
 	while ((*idp)->fti_probe != probe) {
 		idp = &(*idp)->fti_next;
 		ASSERT(*idp != NULL);
 	}
 
 	id = *idp;
 	*idp = id->fti_next;
 	membar_producer();
 
 	ASSERT(id->fti_probe == probe);
 
 	/*
 	 * If there are other registered enablings of this tracepoint, we're
 	 * all done, but if this was the last probe assocated with this
 	 * this tracepoint, we need to remove and free it.
 	 */
 	if (tp->ftt_ids != NULL || tp->ftt_retids != NULL) {
 
 		/*
 		 * If the current probe's tracepoint is in use, swap it
 		 * for an unused tracepoint.
 		 */
 		if (tp == probe->ftp_tps[index].fit_tp) {
 			fasttrap_probe_t *tmp_probe;
 			fasttrap_tracepoint_t **tmp_tp;
 			uint_t tmp_index;
 
 			if (tp->ftt_ids != NULL) {
 				tmp_probe = tp->ftt_ids->fti_probe;
 				/* LINTED - alignment */
 				tmp_index = FASTTRAP_ID_INDEX(tp->ftt_ids);
 				tmp_tp = &tmp_probe->ftp_tps[tmp_index].fit_tp;
 			} else {
 				tmp_probe = tp->ftt_retids->fti_probe;
 				/* LINTED - alignment */
 				tmp_index = FASTTRAP_ID_INDEX(tp->ftt_retids);
 				tmp_tp = &tmp_probe->ftp_tps[tmp_index].fit_tp;
 			}
 
 			ASSERT(*tmp_tp != NULL);
 			ASSERT(*tmp_tp != probe->ftp_tps[index].fit_tp);
 			ASSERT((*tmp_tp)->ftt_ids == NULL);
 			ASSERT((*tmp_tp)->ftt_retids == NULL);
 
 			probe->ftp_tps[index].fit_tp = *tmp_tp;
 			*tmp_tp = tp;
 		}
 
 		mutex_exit(&bucket->ftb_mtx);
 
 		/*
 		 * Tag the modified probe with the generation in which it was
 		 * changed.
 		 */
 		probe->ftp_gen = fasttrap_mod_gen;
 		return;
 	}
 
 	mutex_exit(&bucket->ftb_mtx);
 
 	/*
 	 * We can't safely remove the tracepoint from the set of active
 	 * tracepoints until we've actually removed the fasttrap instruction
 	 * from the process's text. We can, however, operate on this
 	 * tracepoint secure in the knowledge that no other thread is going to
 	 * be looking at it since we hold P_PR_LOCK on the process if it's
 	 * live or we hold the provider lock on the process if it's dead and
 	 * gone.
 	 */
 
 	/*
 	 * We only need to remove the actual instruction if we're looking
 	 * at an existing process
 	 */
 	if (p != NULL) {
 		/*
 		 * If we fail to restore the instruction we need to kill
 		 * this process since it's in a completely unrecoverable
 		 * state.
 		 */
 		if (fasttrap_tracepoint_remove(p, tp) != 0)
 			fasttrap_sigtrap(p, NULL, pc);
 
 		/*
 		 * Decrement the count of the number of tracepoints active
 		 * in the victim process.
 		 */
 #ifdef illumos
 		ASSERT(p->p_proc_flag & P_PR_LOCK);
 #endif
 		p->p_dtrace_count--;
 	}
 
 	/*
 	 * Remove the probe from the hash table of active tracepoints.
 	 */
 	mutex_enter(&bucket->ftb_mtx);
 	pp = (fasttrap_tracepoint_t **)&bucket->ftb_data;
 	ASSERT(*pp != NULL);
 	while (*pp != tp) {
 		pp = &(*pp)->ftt_next;
 		ASSERT(*pp != NULL);
 	}
 
 	*pp = tp->ftt_next;
 	membar_producer();
 
 	mutex_exit(&bucket->ftb_mtx);
 
 	/*
 	 * Tag the modified probe with the generation in which it was changed.
 	 */
 	probe->ftp_gen = fasttrap_mod_gen;
 }
 
 static void
 fasttrap_enable_callbacks(void)
 {
 	/*
 	 * We don't have to play the rw lock game here because we're
 	 * providing something rather than taking something away --
 	 * we can be sure that no threads have tried to follow this
 	 * function pointer yet.
 	 */
 	mutex_enter(&fasttrap_count_mtx);
 	if (fasttrap_pid_count == 0) {
 		ASSERT(dtrace_pid_probe_ptr == NULL);
 		ASSERT(dtrace_return_probe_ptr == NULL);
 		dtrace_pid_probe_ptr = &fasttrap_pid_probe;
 		dtrace_return_probe_ptr = &fasttrap_return_probe;
 	}
 	ASSERT(dtrace_pid_probe_ptr == &fasttrap_pid_probe);
 	ASSERT(dtrace_return_probe_ptr == &fasttrap_return_probe);
 	fasttrap_pid_count++;
 	mutex_exit(&fasttrap_count_mtx);
 }
 
 static void
 fasttrap_disable_callbacks(void)
 {
 #ifdef illumos
 	ASSERT(MUTEX_HELD(&cpu_lock));
 #endif
 
 
 	mutex_enter(&fasttrap_count_mtx);
 	ASSERT(fasttrap_pid_count > 0);
 	fasttrap_pid_count--;
 	if (fasttrap_pid_count == 0) {
 #ifdef illumos
 		cpu_t *cur, *cpu = CPU;
 
 		for (cur = cpu->cpu_next_onln; cur != cpu;
 		    cur = cur->cpu_next_onln) {
 			rw_enter(&cur->cpu_ft_lock, RW_WRITER);
 		}
 #endif
 		dtrace_pid_probe_ptr = NULL;
 		dtrace_return_probe_ptr = NULL;
 #ifdef illumos
 		for (cur = cpu->cpu_next_onln; cur != cpu;
 		    cur = cur->cpu_next_onln) {
 			rw_exit(&cur->cpu_ft_lock);
 		}
 #endif
 	}
 	mutex_exit(&fasttrap_count_mtx);
 }
 
 /*ARGSUSED*/
 static void
 fasttrap_pid_enable(void *arg, dtrace_id_t id, void *parg)
 {
 	fasttrap_probe_t *probe = parg;
 	proc_t *p = NULL;
 	int i, rc;
 
 	ASSERT(probe != NULL);
 	ASSERT(!probe->ftp_enabled);
 	ASSERT(id == probe->ftp_id);
 #ifdef illumos
 	ASSERT(MUTEX_HELD(&cpu_lock));
 #endif
 
 	/*
 	 * Increment the count of enabled probes on this probe's provider;
 	 * the provider can't go away while the probe still exists. We
 	 * must increment this even if we aren't able to properly enable
 	 * this probe.
 	 */
 	mutex_enter(&probe->ftp_prov->ftp_mtx);
 	probe->ftp_prov->ftp_rcount++;
 	mutex_exit(&probe->ftp_prov->ftp_mtx);
 
 	/*
 	 * If this probe's provider is retired (meaning it was valid in a
 	 * previously exec'ed incarnation of this address space), bail out. The
 	 * provider can't go away while we're in this code path.
 	 */
 	if (probe->ftp_prov->ftp_retired)
 		return;
 
 	/*
 	 * If we can't find the process, it may be that we're in the context of
 	 * a fork in which the traced process is being born and we're copying
 	 * USDT probes. Otherwise, the process is gone so bail.
 	 */
 #ifdef illumos
 	if ((p = sprlock(probe->ftp_pid)) == NULL) {
 		if ((curproc->p_flag & SFORKING) == 0)
 			return;
 
 		mutex_enter(&pidlock);
 		p = prfind(probe->ftp_pid);
 
 		/*
 		 * Confirm that curproc is indeed forking the process in which
 		 * we're trying to enable probes.
 		 */
 		ASSERT(p != NULL);
 		ASSERT(p->p_parent == curproc);
 		ASSERT(p->p_stat == SIDL);
 
 		mutex_enter(&p->p_lock);
 		mutex_exit(&pidlock);
 
 		sprlock_proc(p);
 	}
 
 	ASSERT(!(p->p_flag & SVFORK));
 	mutex_exit(&p->p_lock);
 #else
 	if ((p = pfind(probe->ftp_pid)) == NULL)
 		return;
 #endif
 
 	/*
 	 * We have to enable the trap entry point before any user threads have
 	 * the chance to execute the trap instruction we're about to place
 	 * in their process's text.
 	 */
 #ifdef __FreeBSD__
 	/*
 	 * pfind() returns a locked process.
 	 */
 	_PHOLD(p);
 	PROC_UNLOCK(p);
 #endif
 	fasttrap_enable_callbacks();
 
 	/*
 	 * Enable all the tracepoints and add this probe's id to each
 	 * tracepoint's list of active probes.
 	 */
 	for (i = 0; i < probe->ftp_ntps; i++) {
 		if ((rc = fasttrap_tracepoint_enable(p, probe, i)) != 0) {
 			/*
 			 * If enabling the tracepoint failed completely,
 			 * we don't have to disable it; if the failure
 			 * was only partial we must disable it.
 			 */
 			if (rc == FASTTRAP_ENABLE_FAIL)
 				i--;
 			else
 				ASSERT(rc == FASTTRAP_ENABLE_PARTIAL);
 
 			/*
 			 * Back up and pull out all the tracepoints we've
 			 * created so far for this probe.
 			 */
 			while (i >= 0) {
 				fasttrap_tracepoint_disable(p, probe, i);
 				i--;
 			}
 
 #ifdef illumos
 			mutex_enter(&p->p_lock);
 			sprunlock(p);
 #else
 			PRELE(p);
 #endif
 
 			/*
 			 * Since we're not actually enabling this probe,
 			 * drop our reference on the trap table entry.
 			 */
 			fasttrap_disable_callbacks();
 			return;
 		}
 	}
 #ifdef illumos
 	mutex_enter(&p->p_lock);
 	sprunlock(p);
 #else
 	PRELE(p);
 #endif
 
 	probe->ftp_enabled = 1;
 }
 
 /*ARGSUSED*/
 static void
 fasttrap_pid_disable(void *arg, dtrace_id_t id, void *parg)
 {
 	fasttrap_probe_t *probe = parg;
 	fasttrap_provider_t *provider = probe->ftp_prov;
 	proc_t *p;
 	int i, whack = 0;
 
 	ASSERT(id == probe->ftp_id);
 
 	mutex_enter(&provider->ftp_mtx);
 
 	/*
 	 * We won't be able to acquire a /proc-esque lock on the process
 	 * iff the process is dead and gone. In this case, we rely on the
 	 * provider lock as a point of mutual exclusion to prevent other
 	 * DTrace consumers from disabling this probe.
 	 */
 	if ((p = pfind(probe->ftp_pid)) != NULL) {
 #ifdef __FreeBSD__
-		_PHOLD(p);
-		PROC_UNLOCK(p);
+		if (p->p_flag & P_WEXIT) {
+			PROC_UNLOCK(p);
+			p = NULL;
+		} else {
+			_PHOLD(p);
+			PROC_UNLOCK(p);
+		}
 #endif
 	}
 
 	/*
 	 * Disable all the associated tracepoints (for fully enabled probes).
 	 */
 	if (probe->ftp_enabled) {
 		for (i = 0; i < probe->ftp_ntps; i++) {
 			fasttrap_tracepoint_disable(p, probe, i);
 		}
 	}
 
 	ASSERT(provider->ftp_rcount > 0);
 	provider->ftp_rcount--;
 
 	if (p != NULL) {
 		/*
 		 * Even though we may not be able to remove it entirely, we
 		 * mark this retired provider to get a chance to remove some
 		 * of the associated probes.
 		 */
 		if (provider->ftp_retired && !provider->ftp_marked)
 			whack = provider->ftp_marked = 1;
 		mutex_exit(&provider->ftp_mtx);
 	} else {
 		/*
 		 * If the process is dead, we're just waiting for the
 		 * last probe to be disabled to be able to free it.
 		 */
 		if (provider->ftp_rcount == 0 && !provider->ftp_marked)
 			whack = provider->ftp_marked = 1;
 		mutex_exit(&provider->ftp_mtx);
 	}
 
 	if (whack)
 		fasttrap_pid_cleanup();
 
 #ifdef __FreeBSD__
 	if (p != NULL)
 		PRELE(p);
 #endif
 	if (!probe->ftp_enabled)
 		return;
 
 	probe->ftp_enabled = 0;
 
 #ifdef illumos
 	ASSERT(MUTEX_HELD(&cpu_lock));
 #endif
 	fasttrap_disable_callbacks();
 }
 
 /*ARGSUSED*/
 static void
 fasttrap_pid_getargdesc(void *arg, dtrace_id_t id, void *parg,
     dtrace_argdesc_t *desc)
 {
 	fasttrap_probe_t *probe = parg;
 	char *str;
 	int i, ndx;
 
 	desc->dtargd_native[0] = '\0';
 	desc->dtargd_xlate[0] = '\0';
 
 	if (probe->ftp_prov->ftp_retired != 0 ||
 	    desc->dtargd_ndx >= probe->ftp_nargs) {
 		desc->dtargd_ndx = DTRACE_ARGNONE;
 		return;
 	}
 
 	ndx = (probe->ftp_argmap != NULL) ?
 	    probe->ftp_argmap[desc->dtargd_ndx] : desc->dtargd_ndx;
 
 	str = probe->ftp_ntypes;
 	for (i = 0; i < ndx; i++) {
 		str += strlen(str) + 1;
 	}
 
 	ASSERT(strlen(str + 1) < sizeof (desc->dtargd_native));
 	(void) strcpy(desc->dtargd_native, str);
 
 	if (probe->ftp_xtypes == NULL)
 		return;
 
 	str = probe->ftp_xtypes;
 	for (i = 0; i < desc->dtargd_ndx; i++) {
 		str += strlen(str) + 1;
 	}
 
 	ASSERT(strlen(str + 1) < sizeof (desc->dtargd_xlate));
 	(void) strcpy(desc->dtargd_xlate, str);
 }
 
 /*ARGSUSED*/
 static void
 fasttrap_pid_destroy(void *arg, dtrace_id_t id, void *parg)
 {
 	fasttrap_probe_t *probe = parg;
 	int i;
 	size_t size;
 
 	ASSERT(probe != NULL);
 	ASSERT(!probe->ftp_enabled);
 	ASSERT(fasttrap_total >= probe->ftp_ntps);
 
 	atomic_add_32(&fasttrap_total, -probe->ftp_ntps);
 	size = offsetof(fasttrap_probe_t, ftp_tps[probe->ftp_ntps]);
 
 	if (probe->ftp_gen + 1 >= fasttrap_mod_gen)
 		fasttrap_mod_barrier(probe->ftp_gen);
 
 	for (i = 0; i < probe->ftp_ntps; i++) {
 		kmem_free(probe->ftp_tps[i].fit_tp,
 		    sizeof (fasttrap_tracepoint_t));
 	}
 
 	kmem_free(probe, size);
 }
 
 
 static const dtrace_pattr_t pid_attr = {
 { DTRACE_STABILITY_EVOLVING, DTRACE_STABILITY_EVOLVING, DTRACE_CLASS_ISA },
 { DTRACE_STABILITY_PRIVATE, DTRACE_STABILITY_PRIVATE, DTRACE_CLASS_UNKNOWN },
 { DTRACE_STABILITY_PRIVATE, DTRACE_STABILITY_PRIVATE, DTRACE_CLASS_UNKNOWN },
 { DTRACE_STABILITY_EVOLVING, DTRACE_STABILITY_EVOLVING, DTRACE_CLASS_ISA },
 { DTRACE_STABILITY_PRIVATE, DTRACE_STABILITY_PRIVATE, DTRACE_CLASS_UNKNOWN },
 };
 
 static dtrace_pops_t pid_pops = {
 	fasttrap_pid_provide,
 	NULL,
 	fasttrap_pid_enable,
 	fasttrap_pid_disable,
 	NULL,
 	NULL,
 	fasttrap_pid_getargdesc,
 	fasttrap_pid_getarg,
 	NULL,
 	fasttrap_pid_destroy
 };
 
 static dtrace_pops_t usdt_pops = {
 	fasttrap_pid_provide,
 	NULL,
 	fasttrap_pid_enable,
 	fasttrap_pid_disable,
 	NULL,
 	NULL,
 	fasttrap_pid_getargdesc,
 	fasttrap_usdt_getarg,
 	NULL,
 	fasttrap_pid_destroy
 };
 
 static fasttrap_proc_t *
 fasttrap_proc_lookup(pid_t pid)
 {
 	fasttrap_bucket_t *bucket;
 	fasttrap_proc_t *fprc, *new_fprc;
 
 
 	bucket = &fasttrap_procs.fth_table[FASTTRAP_PROCS_INDEX(pid)];
 	mutex_enter(&bucket->ftb_mtx);
 
 	for (fprc = bucket->ftb_data; fprc != NULL; fprc = fprc->ftpc_next) {
 		if (fprc->ftpc_pid == pid && fprc->ftpc_acount != 0) {
 			mutex_enter(&fprc->ftpc_mtx);
 			mutex_exit(&bucket->ftb_mtx);
 			fprc->ftpc_rcount++;
 			atomic_inc_64(&fprc->ftpc_acount);
 			ASSERT(fprc->ftpc_acount <= fprc->ftpc_rcount);
 			mutex_exit(&fprc->ftpc_mtx);
 
 			return (fprc);
 		}
 	}
 
 	/*
 	 * Drop the bucket lock so we don't try to perform a sleeping
 	 * allocation under it.
 	 */
 	mutex_exit(&bucket->ftb_mtx);
 
 	new_fprc = kmem_zalloc(sizeof (fasttrap_proc_t), KM_SLEEP);
 	new_fprc->ftpc_pid = pid;
 	new_fprc->ftpc_rcount = 1;
 	new_fprc->ftpc_acount = 1;
 #ifndef illumos
 	mutex_init(&new_fprc->ftpc_mtx, "fasttrap proc mtx", MUTEX_DEFAULT,
 	    NULL);
 #endif
 
 	mutex_enter(&bucket->ftb_mtx);
 
 	/*
 	 * Take another lap through the list to make sure a proc hasn't
 	 * been created for this pid while we weren't under the bucket lock.
 	 */
 	for (fprc = bucket->ftb_data; fprc != NULL; fprc = fprc->ftpc_next) {
 		if (fprc->ftpc_pid == pid && fprc->ftpc_acount != 0) {
 			mutex_enter(&fprc->ftpc_mtx);
 			mutex_exit(&bucket->ftb_mtx);
 			fprc->ftpc_rcount++;
 			atomic_inc_64(&fprc->ftpc_acount);
 			ASSERT(fprc->ftpc_acount <= fprc->ftpc_rcount);
 			mutex_exit(&fprc->ftpc_mtx);
 
 			kmem_free(new_fprc, sizeof (fasttrap_proc_t));
 
 			return (fprc);
 		}
 	}
 
 	new_fprc->ftpc_next = bucket->ftb_data;
 	bucket->ftb_data = new_fprc;
 
 	mutex_exit(&bucket->ftb_mtx);
 
 	return (new_fprc);
 }
 
 static void
 fasttrap_proc_release(fasttrap_proc_t *proc)
 {
 	fasttrap_bucket_t *bucket;
 	fasttrap_proc_t *fprc, **fprcp;
 	pid_t pid = proc->ftpc_pid;
 #ifndef illumos
 	fasttrap_scrblock_t *scrblk, *scrblktmp;
 	fasttrap_scrspace_t *scrspc, *scrspctmp;
 	struct proc *p;
 	struct thread *td;
 #endif
 
 	mutex_enter(&proc->ftpc_mtx);
 
 	ASSERT(proc->ftpc_rcount != 0);
 	ASSERT(proc->ftpc_acount <= proc->ftpc_rcount);
 
 	if (--proc->ftpc_rcount != 0) {
 		mutex_exit(&proc->ftpc_mtx);
 		return;
 	}
 
 #ifndef illumos
 	/*
 	 * Free all structures used to manage per-thread scratch space.
 	 */
 	LIST_FOREACH_SAFE(scrblk, &proc->ftpc_scrblks, ftsb_next,
 	    scrblktmp) {
 		LIST_REMOVE(scrblk, ftsb_next);
 		free(scrblk, M_SOLARIS);
 	}
 	LIST_FOREACH_SAFE(scrspc, &proc->ftpc_fscr, ftss_next, scrspctmp) {
 		LIST_REMOVE(scrspc, ftss_next);
 		free(scrspc, M_SOLARIS);
 	}
 	LIST_FOREACH_SAFE(scrspc, &proc->ftpc_ascr, ftss_next, scrspctmp) {
 		LIST_REMOVE(scrspc, ftss_next);
 		free(scrspc, M_SOLARIS);
 	}
 
 	if ((p = pfind(pid)) != NULL) {
 		FOREACH_THREAD_IN_PROC(p, td)
 			td->t_dtrace_sscr = NULL;
 		PROC_UNLOCK(p);
 	}
 #endif
 
 	mutex_exit(&proc->ftpc_mtx);
 
 	/*
 	 * There should definitely be no live providers associated with this
 	 * process at this point.
 	 */
 	ASSERT(proc->ftpc_acount == 0);
 
 	bucket = &fasttrap_procs.fth_table[FASTTRAP_PROCS_INDEX(pid)];
 	mutex_enter(&bucket->ftb_mtx);
 
 	fprcp = (fasttrap_proc_t **)&bucket->ftb_data;
 	while ((fprc = *fprcp) != NULL) {
 		if (fprc == proc)
 			break;
 
 		fprcp = &fprc->ftpc_next;
 	}
 
 	/*
 	 * Something strange has happened if we can't find the proc.
 	 */
 	ASSERT(fprc != NULL);
 
 	*fprcp = fprc->ftpc_next;
 
 	mutex_exit(&bucket->ftb_mtx);
 
 	kmem_free(fprc, sizeof (fasttrap_proc_t));
 }
 
 /*
  * Lookup a fasttrap-managed provider based on its name and associated pid.
  * If the pattr argument is non-NULL, this function instantiates the provider
  * if it doesn't exist otherwise it returns NULL. The provider is returned
  * with its lock held.
  */
 static fasttrap_provider_t *
 fasttrap_provider_lookup(pid_t pid, const char *name,
     const dtrace_pattr_t *pattr)
 {
 	fasttrap_provider_t *fp, *new_fp = NULL;
 	fasttrap_bucket_t *bucket;
 	char provname[DTRACE_PROVNAMELEN];
 	proc_t *p;
 	cred_t *cred;
 
 	ASSERT(strlen(name) < sizeof (fp->ftp_name));
 	ASSERT(pattr != NULL);
 
 	bucket = &fasttrap_provs.fth_table[FASTTRAP_PROVS_INDEX(pid, name)];
 	mutex_enter(&bucket->ftb_mtx);
 
 	/*
 	 * Take a lap through the list and return the match if we find it.
 	 */
 	for (fp = bucket->ftb_data; fp != NULL; fp = fp->ftp_next) {
 		if (fp->ftp_pid == pid && strcmp(fp->ftp_name, name) == 0 &&
 		    !fp->ftp_retired) {
 			mutex_enter(&fp->ftp_mtx);
 			mutex_exit(&bucket->ftb_mtx);
 			return (fp);
 		}
 	}
 
 	/*
 	 * Drop the bucket lock so we don't try to perform a sleeping
 	 * allocation under it.
 	 */
 	mutex_exit(&bucket->ftb_mtx);
 
 	/*
 	 * Make sure the process exists, isn't a child created as the result
 	 * of a vfork(2), and isn't a zombie (but may be in fork).
 	 */
 	if ((p = pfind(pid)) == NULL)
 		return (NULL);
 
 	/*
 	 * Increment p_dtrace_probes so that the process knows to inform us
 	 * when it exits or execs. fasttrap_provider_free() decrements this
 	 * when we're done with this provider.
 	 */
 	p->p_dtrace_probes++;
 
 	/*
 	 * Grab the credentials for this process so we have
 	 * something to pass to dtrace_register().
 	 */
 	PROC_LOCK_ASSERT(p, MA_OWNED);
 	crhold(p->p_ucred);
 	cred = p->p_ucred;
 	PROC_UNLOCK(p);
 
 	new_fp = kmem_zalloc(sizeof (fasttrap_provider_t), KM_SLEEP);
 	new_fp->ftp_pid = pid;
 	new_fp->ftp_proc = fasttrap_proc_lookup(pid);
 #ifndef illumos
 	mutex_init(&new_fp->ftp_mtx, "provider mtx", MUTEX_DEFAULT, NULL);
 	mutex_init(&new_fp->ftp_cmtx, "lock on creating", MUTEX_DEFAULT, NULL);
 #endif
 
 	ASSERT(new_fp->ftp_proc != NULL);
 
 	mutex_enter(&bucket->ftb_mtx);
 
 	/*
 	 * Take another lap through the list to make sure a provider hasn't
 	 * been created for this pid while we weren't under the bucket lock.
 	 */
 	for (fp = bucket->ftb_data; fp != NULL; fp = fp->ftp_next) {
 		if (fp->ftp_pid == pid && strcmp(fp->ftp_name, name) == 0 &&
 		    !fp->ftp_retired) {
 			mutex_enter(&fp->ftp_mtx);
 			mutex_exit(&bucket->ftb_mtx);
 			fasttrap_provider_free(new_fp);
 			crfree(cred);
 			return (fp);
 		}
 	}
 
 	(void) strcpy(new_fp->ftp_name, name);
 
 	/*
 	 * Fail and return NULL if either the provider name is too long
 	 * or we fail to register this new provider with the DTrace
 	 * framework. Note that this is the only place we ever construct
 	 * the full provider name -- we keep it in pieces in the provider
 	 * structure.
 	 */
 	if (snprintf(provname, sizeof (provname), "%s%u", name, (uint_t)pid) >=
 	    sizeof (provname) ||
 	    dtrace_register(provname, pattr,
 	    DTRACE_PRIV_PROC | DTRACE_PRIV_OWNER | DTRACE_PRIV_ZONEOWNER, cred,
 	    pattr == &pid_attr ? &pid_pops : &usdt_pops, new_fp,
 	    &new_fp->ftp_provid) != 0) {
 		mutex_exit(&bucket->ftb_mtx);
 		fasttrap_provider_free(new_fp);
 		crfree(cred);
 		return (NULL);
 	}
 
 	new_fp->ftp_next = bucket->ftb_data;
 	bucket->ftb_data = new_fp;
 
 	mutex_enter(&new_fp->ftp_mtx);
 	mutex_exit(&bucket->ftb_mtx);
 
 	crfree(cred);
 	return (new_fp);
 }
 
 static void
 fasttrap_provider_free(fasttrap_provider_t *provider)
 {
 	pid_t pid = provider->ftp_pid;
 	proc_t *p;
 
 	/*
 	 * There need to be no associated enabled probes, no consumers
 	 * creating probes, and no meta providers referencing this provider.
 	 */
 	ASSERT(provider->ftp_rcount == 0);
 	ASSERT(provider->ftp_ccount == 0);
 	ASSERT(provider->ftp_mcount == 0);
 
 	/*
 	 * If this provider hasn't been retired, we need to explicitly drop the
 	 * count of active providers on the associated process structure.
 	 */
 	if (!provider->ftp_retired) {
 		atomic_dec_64(&provider->ftp_proc->ftpc_acount);
 		ASSERT(provider->ftp_proc->ftpc_acount <
 		    provider->ftp_proc->ftpc_rcount);
 	}
 
 	fasttrap_proc_release(provider->ftp_proc);
 
 #ifndef illumos
 	mutex_destroy(&provider->ftp_mtx);
 	mutex_destroy(&provider->ftp_cmtx);
 #endif
 	kmem_free(provider, sizeof (fasttrap_provider_t));
 
 	/*
 	 * Decrement p_dtrace_probes on the process whose provider we're
 	 * freeing. We don't have to worry about clobbering somone else's
 	 * modifications to it because we have locked the bucket that
 	 * corresponds to this process's hash chain in the provider hash
 	 * table. Don't sweat it if we can't find the process.
 	 */
 	if ((p = pfind(pid)) == NULL) {
 		return;
 	}
 
 	p->p_dtrace_probes--;
 #ifndef illumos
 	PROC_UNLOCK(p);
 #endif
 }
 
 static void
 fasttrap_provider_retire(pid_t pid, const char *name, int mprov)
 {
 	fasttrap_provider_t *fp;
 	fasttrap_bucket_t *bucket;
 	dtrace_provider_id_t provid;
 
 	ASSERT(strlen(name) < sizeof (fp->ftp_name));
 
 	bucket = &fasttrap_provs.fth_table[FASTTRAP_PROVS_INDEX(pid, name)];
 	mutex_enter(&bucket->ftb_mtx);
 
 	for (fp = bucket->ftb_data; fp != NULL; fp = fp->ftp_next) {
 		if (fp->ftp_pid == pid && strcmp(fp->ftp_name, name) == 0 &&
 		    !fp->ftp_retired)
 			break;
 	}
 
 	if (fp == NULL) {
 		mutex_exit(&bucket->ftb_mtx);
 		return;
 	}
 
 	mutex_enter(&fp->ftp_mtx);
 	ASSERT(!mprov || fp->ftp_mcount > 0);
 	if (mprov && --fp->ftp_mcount != 0)  {
 		mutex_exit(&fp->ftp_mtx);
 		mutex_exit(&bucket->ftb_mtx);
 		return;
 	}
 
 	/*
 	 * Mark the provider to be removed in our post-processing step, mark it
 	 * retired, and drop the active count on its proc. Marking it indicates
 	 * that we should try to remove it; setting the retired flag indicates
 	 * that we're done with this provider; dropping the active the proc
 	 * releases our hold, and when this reaches zero (as it will during
 	 * exit or exec) the proc and associated providers become defunct.
 	 *
 	 * We obviously need to take the bucket lock before the provider lock
 	 * to perform the lookup, but we need to drop the provider lock
 	 * before calling into the DTrace framework since we acquire the
 	 * provider lock in callbacks invoked from the DTrace framework. The
 	 * bucket lock therefore protects the integrity of the provider hash
 	 * table.
 	 */
 	atomic_dec_64(&fp->ftp_proc->ftpc_acount);
 	ASSERT(fp->ftp_proc->ftpc_acount < fp->ftp_proc->ftpc_rcount);
 
 	fp->ftp_retired = 1;
 	fp->ftp_marked = 1;
 	provid = fp->ftp_provid;
 	mutex_exit(&fp->ftp_mtx);
 
 	/*
 	 * We don't have to worry about invalidating the same provider twice
 	 * since fasttrap_provider_lookup() will ignore provider that have
 	 * been marked as retired.
 	 */
 	dtrace_invalidate(provid);
 
 	mutex_exit(&bucket->ftb_mtx);
 
 	fasttrap_pid_cleanup();
 }
 
 static int
 fasttrap_uint32_cmp(const void *ap, const void *bp)
 {
 	return (*(const uint32_t *)ap - *(const uint32_t *)bp);
 }
 
 static int
 fasttrap_uint64_cmp(const void *ap, const void *bp)
 {
 	return (*(const uint64_t *)ap - *(const uint64_t *)bp);
 }
 
 static int
 fasttrap_add_probe(fasttrap_probe_spec_t *pdata)
 {
 	fasttrap_provider_t *provider;
 	fasttrap_probe_t *pp;
 	fasttrap_tracepoint_t *tp;
 	char *name;
 	int i, aframes = 0, whack;
 
 	/*
 	 * There needs to be at least one desired trace point.
 	 */
 	if (pdata->ftps_noffs == 0)
 		return (EINVAL);
 
 	switch (pdata->ftps_type) {
 	case DTFTP_ENTRY:
 		name = "entry";
 		aframes = FASTTRAP_ENTRY_AFRAMES;
 		break;
 	case DTFTP_RETURN:
 		name = "return";
 		aframes = FASTTRAP_RETURN_AFRAMES;
 		break;
 	case DTFTP_OFFSETS:
 		name = NULL;
 		break;
 	default:
 		return (EINVAL);
 	}
 
 	if ((provider = fasttrap_provider_lookup(pdata->ftps_pid,
 	    FASTTRAP_PID_NAME, &pid_attr)) == NULL)
 		return (ESRCH);
 
 	/*
 	 * Increment this reference count to indicate that a consumer is
 	 * actively adding a new probe associated with this provider. This
 	 * prevents the provider from being deleted -- we'll need to check
 	 * for pending deletions when we drop this reference count.
 	 */
 	provider->ftp_ccount++;
 	mutex_exit(&provider->ftp_mtx);
 
 	/*
 	 * Grab the creation lock to ensure consistency between calls to
 	 * dtrace_probe_lookup() and dtrace_probe_create() in the face of
 	 * other threads creating probes. We must drop the provider lock
 	 * before taking this lock to avoid a three-way deadlock with the
 	 * DTrace framework.
 	 */
 	mutex_enter(&provider->ftp_cmtx);
 
 	if (name == NULL) {
 		for (i = 0; i < pdata->ftps_noffs; i++) {
 			char name_str[17];
 
 			(void) sprintf(name_str, "%llx",
 			    (unsigned long long)pdata->ftps_offs[i]);
 
 			if (dtrace_probe_lookup(provider->ftp_provid,
 			    pdata->ftps_mod, pdata->ftps_func, name_str) != 0)
 				continue;
 
 			atomic_inc_32(&fasttrap_total);
 
 			if (fasttrap_total > fasttrap_max) {
 				atomic_dec_32(&fasttrap_total);
 				goto no_mem;
 			}
 
 			pp = kmem_zalloc(sizeof (fasttrap_probe_t), KM_SLEEP);
 
 			pp->ftp_prov = provider;
 			pp->ftp_faddr = pdata->ftps_pc;
 			pp->ftp_fsize = pdata->ftps_size;
 			pp->ftp_pid = pdata->ftps_pid;
 			pp->ftp_ntps = 1;
 
 			tp = kmem_zalloc(sizeof (fasttrap_tracepoint_t),
 			    KM_SLEEP);
 
 			tp->ftt_proc = provider->ftp_proc;
 			tp->ftt_pc = pdata->ftps_offs[i] + pdata->ftps_pc;
 			tp->ftt_pid = pdata->ftps_pid;
 
 			pp->ftp_tps[0].fit_tp = tp;
 			pp->ftp_tps[0].fit_id.fti_probe = pp;
 			pp->ftp_tps[0].fit_id.fti_ptype = pdata->ftps_type;
 
 			pp->ftp_id = dtrace_probe_create(provider->ftp_provid,
 			    pdata->ftps_mod, pdata->ftps_func, name_str,
 			    FASTTRAP_OFFSET_AFRAMES, pp);
 		}
 
 	} else if (dtrace_probe_lookup(provider->ftp_provid, pdata->ftps_mod,
 	    pdata->ftps_func, name) == 0) {
 		atomic_add_32(&fasttrap_total, pdata->ftps_noffs);
 
 		if (fasttrap_total > fasttrap_max) {
 			atomic_add_32(&fasttrap_total, -pdata->ftps_noffs);
 			goto no_mem;
 		}
 
 		/*
 		 * Make sure all tracepoint program counter values are unique.
 		 * We later assume that each probe has exactly one tracepoint
 		 * for a given pc.
 		 */
 		qsort(pdata->ftps_offs, pdata->ftps_noffs,
 		    sizeof (uint64_t), fasttrap_uint64_cmp);
 		for (i = 1; i < pdata->ftps_noffs; i++) {
 			if (pdata->ftps_offs[i] > pdata->ftps_offs[i - 1])
 				continue;
 
 			atomic_add_32(&fasttrap_total, -pdata->ftps_noffs);
 			goto no_mem;
 		}
 
 		ASSERT(pdata->ftps_noffs > 0);
 		pp = kmem_zalloc(offsetof(fasttrap_probe_t,
 		    ftp_tps[pdata->ftps_noffs]), KM_SLEEP);
 
 		pp->ftp_prov = provider;
 		pp->ftp_faddr = pdata->ftps_pc;
 		pp->ftp_fsize = pdata->ftps_size;
 		pp->ftp_pid = pdata->ftps_pid;
 		pp->ftp_ntps = pdata->ftps_noffs;
 
 		for (i = 0; i < pdata->ftps_noffs; i++) {
 			tp = kmem_zalloc(sizeof (fasttrap_tracepoint_t),
 			    KM_SLEEP);
 
 			tp->ftt_proc = provider->ftp_proc;
 			tp->ftt_pc = pdata->ftps_offs[i] + pdata->ftps_pc;
 			tp->ftt_pid = pdata->ftps_pid;
 
 			pp->ftp_tps[i].fit_tp = tp;
 			pp->ftp_tps[i].fit_id.fti_probe = pp;
 			pp->ftp_tps[i].fit_id.fti_ptype = pdata->ftps_type;
 		}
 
 		pp->ftp_id = dtrace_probe_create(provider->ftp_provid,
 		    pdata->ftps_mod, pdata->ftps_func, name, aframes, pp);
 	}
 
 	mutex_exit(&provider->ftp_cmtx);
 
 	/*
 	 * We know that the provider is still valid since we incremented the
 	 * creation reference count. If someone tried to clean up this provider
 	 * while we were using it (e.g. because the process called exec(2) or
 	 * exit(2)), take note of that and try to clean it up now.
 	 */
 	mutex_enter(&provider->ftp_mtx);
 	provider->ftp_ccount--;
 	whack = provider->ftp_retired;
 	mutex_exit(&provider->ftp_mtx);
 
 	if (whack)
 		fasttrap_pid_cleanup();
 
 	return (0);
 
 no_mem:
 	/*
 	 * If we've exhausted the allowable resources, we'll try to remove
 	 * this provider to free some up. This is to cover the case where
 	 * the user has accidentally created many more probes than was
 	 * intended (e.g. pid123:::).
 	 */
 	mutex_exit(&provider->ftp_cmtx);
 	mutex_enter(&provider->ftp_mtx);
 	provider->ftp_ccount--;
 	provider->ftp_marked = 1;
 	mutex_exit(&provider->ftp_mtx);
 
 	fasttrap_pid_cleanup();
 
 	return (ENOMEM);
 }
 
 /*ARGSUSED*/
 static void *
 fasttrap_meta_provide(void *arg, dtrace_helper_provdesc_t *dhpv, pid_t pid)
 {
 	fasttrap_provider_t *provider;
 
 	/*
 	 * A 32-bit unsigned integer (like a pid for example) can be
 	 * expressed in 10 or fewer decimal digits. Make sure that we'll
 	 * have enough space for the provider name.
 	 */
 	if (strlen(dhpv->dthpv_provname) + 10 >=
 	    sizeof (provider->ftp_name)) {
 		printf("failed to instantiate provider %s: "
 		    "name too long to accomodate pid", dhpv->dthpv_provname);
 		return (NULL);
 	}
 
 	/*
 	 * Don't let folks spoof the true pid provider.
 	 */
 	if (strcmp(dhpv->dthpv_provname, FASTTRAP_PID_NAME) == 0) {
 		printf("failed to instantiate provider %s: "
 		    "%s is an invalid name", dhpv->dthpv_provname,
 		    FASTTRAP_PID_NAME);
 		return (NULL);
 	}
 
 	/*
 	 * The highest stability class that fasttrap supports is ISA; cap
 	 * the stability of the new provider accordingly.
 	 */
 	if (dhpv->dthpv_pattr.dtpa_provider.dtat_class > DTRACE_CLASS_ISA)
 		dhpv->dthpv_pattr.dtpa_provider.dtat_class = DTRACE_CLASS_ISA;
 	if (dhpv->dthpv_pattr.dtpa_mod.dtat_class > DTRACE_CLASS_ISA)
 		dhpv->dthpv_pattr.dtpa_mod.dtat_class = DTRACE_CLASS_ISA;
 	if (dhpv->dthpv_pattr.dtpa_func.dtat_class > DTRACE_CLASS_ISA)
 		dhpv->dthpv_pattr.dtpa_func.dtat_class = DTRACE_CLASS_ISA;
 	if (dhpv->dthpv_pattr.dtpa_name.dtat_class > DTRACE_CLASS_ISA)
 		dhpv->dthpv_pattr.dtpa_name.dtat_class = DTRACE_CLASS_ISA;
 	if (dhpv->dthpv_pattr.dtpa_args.dtat_class > DTRACE_CLASS_ISA)
 		dhpv->dthpv_pattr.dtpa_args.dtat_class = DTRACE_CLASS_ISA;
 
 	if ((provider = fasttrap_provider_lookup(pid, dhpv->dthpv_provname,
 	    &dhpv->dthpv_pattr)) == NULL) {
 		printf("failed to instantiate provider %s for "
 		    "process %u",  dhpv->dthpv_provname, (uint_t)pid);
 		return (NULL);
 	}
 
 	/*
 	 * Up the meta provider count so this provider isn't removed until
 	 * the meta provider has been told to remove it.
 	 */
 	provider->ftp_mcount++;
 
 	mutex_exit(&provider->ftp_mtx);
 
 	return (provider);
 }
 
 /*ARGSUSED*/
 static void
 fasttrap_meta_create_probe(void *arg, void *parg,
     dtrace_helper_probedesc_t *dhpb)
 {
 	fasttrap_provider_t *provider = parg;
 	fasttrap_probe_t *pp;
 	fasttrap_tracepoint_t *tp;
 	int i, j;
 	uint32_t ntps;
 
 	/*
 	 * Since the meta provider count is non-zero we don't have to worry
 	 * about this provider disappearing.
 	 */
 	ASSERT(provider->ftp_mcount > 0);
 
 	/*
 	 * The offsets must be unique.
 	 */
 	qsort(dhpb->dthpb_offs, dhpb->dthpb_noffs, sizeof (uint32_t),
 	    fasttrap_uint32_cmp);
 	for (i = 1; i < dhpb->dthpb_noffs; i++) {
 		if (dhpb->dthpb_base + dhpb->dthpb_offs[i] <=
 		    dhpb->dthpb_base + dhpb->dthpb_offs[i - 1])
 			return;
 	}
 
 	qsort(dhpb->dthpb_enoffs, dhpb->dthpb_nenoffs, sizeof (uint32_t),
 	    fasttrap_uint32_cmp);
 	for (i = 1; i < dhpb->dthpb_nenoffs; i++) {
 		if (dhpb->dthpb_base + dhpb->dthpb_enoffs[i] <=
 		    dhpb->dthpb_base + dhpb->dthpb_enoffs[i - 1])
 			return;
 	}
 
 	/*
 	 * Grab the creation lock to ensure consistency between calls to
 	 * dtrace_probe_lookup() and dtrace_probe_create() in the face of
 	 * other threads creating probes.
 	 */
 	mutex_enter(&provider->ftp_cmtx);
 
 	if (dtrace_probe_lookup(provider->ftp_provid, dhpb->dthpb_mod,
 	    dhpb->dthpb_func, dhpb->dthpb_name) != 0) {
 		mutex_exit(&provider->ftp_cmtx);
 		return;
 	}
 
 	ntps = dhpb->dthpb_noffs + dhpb->dthpb_nenoffs;
 	ASSERT(ntps > 0);
 
 	atomic_add_32(&fasttrap_total, ntps);
 
 	if (fasttrap_total > fasttrap_max) {
 		atomic_add_32(&fasttrap_total, -ntps);
 		mutex_exit(&provider->ftp_cmtx);
 		return;
 	}
 
 	pp = kmem_zalloc(offsetof(fasttrap_probe_t, ftp_tps[ntps]), KM_SLEEP);
 
 	pp->ftp_prov = provider;
 	pp->ftp_pid = provider->ftp_pid;
 	pp->ftp_ntps = ntps;
 	pp->ftp_nargs = dhpb->dthpb_xargc;
 	pp->ftp_xtypes = dhpb->dthpb_xtypes;
 	pp->ftp_ntypes = dhpb->dthpb_ntypes;
 
 	/*
 	 * First create a tracepoint for each actual point of interest.
 	 */
 	for (i = 0; i < dhpb->dthpb_noffs; i++) {
 		tp = kmem_zalloc(sizeof (fasttrap_tracepoint_t), KM_SLEEP);
 
 		tp->ftt_proc = provider->ftp_proc;
 		tp->ftt_pc = dhpb->dthpb_base + dhpb->dthpb_offs[i];
 		tp->ftt_pid = provider->ftp_pid;
 
 		pp->ftp_tps[i].fit_tp = tp;
 		pp->ftp_tps[i].fit_id.fti_probe = pp;
 #ifdef __sparc
 		pp->ftp_tps[i].fit_id.fti_ptype = DTFTP_POST_OFFSETS;
 #else
 		pp->ftp_tps[i].fit_id.fti_ptype = DTFTP_OFFSETS;
 #endif
 	}
 
 	/*
 	 * Then create a tracepoint for each is-enabled point.
 	 */
 	for (j = 0; i < ntps; i++, j++) {
 		tp = kmem_zalloc(sizeof (fasttrap_tracepoint_t), KM_SLEEP);
 
 		tp->ftt_proc = provider->ftp_proc;
 		tp->ftt_pc = dhpb->dthpb_base + dhpb->dthpb_enoffs[j];
 		tp->ftt_pid = provider->ftp_pid;
 
 		pp->ftp_tps[i].fit_tp = tp;
 		pp->ftp_tps[i].fit_id.fti_probe = pp;
 		pp->ftp_tps[i].fit_id.fti_ptype = DTFTP_IS_ENABLED;
 	}
 
 	/*
 	 * If the arguments are shuffled around we set the argument remapping
 	 * table. Later, when the probe fires, we only remap the arguments
 	 * if the table is non-NULL.
 	 */
 	for (i = 0; i < dhpb->dthpb_xargc; i++) {
 		if (dhpb->dthpb_args[i] != i) {
 			pp->ftp_argmap = dhpb->dthpb_args;
 			break;
 		}
 	}
 
 	/*
 	 * The probe is fully constructed -- register it with DTrace.
 	 */
 	pp->ftp_id = dtrace_probe_create(provider->ftp_provid, dhpb->dthpb_mod,
 	    dhpb->dthpb_func, dhpb->dthpb_name, FASTTRAP_OFFSET_AFRAMES, pp);
 
 	mutex_exit(&provider->ftp_cmtx);
 }
 
 /*ARGSUSED*/
 static void
 fasttrap_meta_remove(void *arg, dtrace_helper_provdesc_t *dhpv, pid_t pid)
 {
 	/*
 	 * Clean up the USDT provider. There may be active consumers of the
 	 * provider busy adding probes, no damage will actually befall the
 	 * provider until that count has dropped to zero. This just puts
 	 * the provider on death row.
 	 */
 	fasttrap_provider_retire(pid, dhpv->dthpv_provname, 1);
 }
 
 static dtrace_mops_t fasttrap_mops = {
 	fasttrap_meta_create_probe,
 	fasttrap_meta_provide,
 	fasttrap_meta_remove
 };
 
 /*ARGSUSED*/
 static int
 fasttrap_open(struct cdev *dev __unused, int oflags __unused,
     int devtype __unused, struct thread *td __unused)
 {
 	return (0);
 }
 
 /*ARGSUSED*/
 static int
 fasttrap_ioctl(struct cdev *dev, u_long cmd, caddr_t arg, int fflag,
     struct thread *td)
 {
 #ifdef notyet
 	struct kinfo_proc kp;
 	const cred_t *cr = td->td_ucred;
 #endif
 	if (!dtrace_attached())
 		return (EAGAIN);
 
 	if (cmd == FASTTRAPIOC_MAKEPROBE) {
 		fasttrap_probe_spec_t *uprobe = *(fasttrap_probe_spec_t **)arg;
 		fasttrap_probe_spec_t *probe;
 		uint64_t noffs;
 		size_t size;
 		int ret, err;
 
 		if (copyin(&uprobe->ftps_noffs, &noffs,
 		    sizeof (uprobe->ftps_noffs)))
 			return (EFAULT);
 
 		/*
 		 * Probes must have at least one tracepoint.
 		 */
 		if (noffs == 0)
 			return (EINVAL);
 
 		size = sizeof (fasttrap_probe_spec_t) +
 		    sizeof (probe->ftps_offs[0]) * (noffs - 1);
 
 		if (size > 1024 * 1024)
 			return (ENOMEM);
 
 		probe = kmem_alloc(size, KM_SLEEP);
 
 		if (copyin(uprobe, probe, size) != 0 ||
 		    probe->ftps_noffs != noffs) {
 			kmem_free(probe, size);
 			return (EFAULT);
 		}
 
 		/*
 		 * Verify that the function and module strings contain no
 		 * funny characters.
 		 */
 		if (u8_validate(probe->ftps_func, strlen(probe->ftps_func),
 		    NULL, U8_VALIDATE_ENTIRE, &err) < 0) {
 			ret = EINVAL;
 			goto err;
 		}
 
 		if (u8_validate(probe->ftps_mod, strlen(probe->ftps_mod),
 		    NULL, U8_VALIDATE_ENTIRE, &err) < 0) {
 			ret = EINVAL;
 			goto err;
 		}
 
 #ifdef notyet
 		if (!PRIV_POLICY_CHOICE(cr, PRIV_ALL, B_FALSE)) {
 			proc_t *p;
 			pid_t pid = probe->ftps_pid;
 
 #ifdef illumos
 			mutex_enter(&pidlock);
 #endif
 			/*
 			 * Report an error if the process doesn't exist
 			 * or is actively being birthed.
 			 */
 			sx_slock(&proctree_lock);
 			p = pfind(pid);
 			if (p)
 				fill_kinfo_proc(p, &kp);
 			sx_sunlock(&proctree_lock);
 			if (p == NULL || kp.ki_stat == SIDL) {
 #ifdef illumos
 				mutex_exit(&pidlock);
 #endif
 				return (ESRCH);
 			}
 #ifdef illumos
 			mutex_enter(&p->p_lock);
 			mutex_exit(&pidlock);
 #else
 			PROC_LOCK_ASSERT(p, MA_OWNED);
 #endif
 
 #ifdef notyet
 			if ((ret = priv_proc_cred_perm(cr, p, NULL,
 			    VREAD | VWRITE)) != 0) {
 #ifdef illumos
 				mutex_exit(&p->p_lock);
 #else
 				PROC_UNLOCK(p);
 #endif
 				return (ret);
 			}
 #endif /* notyet */
 #ifdef illumos
 			mutex_exit(&p->p_lock);
 #else
 			PROC_UNLOCK(p);
 #endif
 		}
 #endif /* notyet */
 
 		ret = fasttrap_add_probe(probe);
 err:
 		kmem_free(probe, size);
 
 		return (ret);
 
 	} else if (cmd == FASTTRAPIOC_GETINSTR) {
 		fasttrap_instr_query_t instr;
 		fasttrap_tracepoint_t *tp;
 		uint_t index;
 #ifdef illumos
 		int ret;
 #endif
 
 #ifdef illumos
 		if (copyin((void *)arg, &instr, sizeof (instr)) != 0)
 			return (EFAULT);
 #endif
 
 #ifdef notyet
 		if (!PRIV_POLICY_CHOICE(cr, PRIV_ALL, B_FALSE)) {
 			proc_t *p;
 			pid_t pid = instr.ftiq_pid;
 
 #ifdef illumos
 			mutex_enter(&pidlock);
 #endif
 			/*
 			 * Report an error if the process doesn't exist
 			 * or is actively being birthed.
 			 */
 			sx_slock(&proctree_lock);
 			p = pfind(pid);
 			if (p)
 				fill_kinfo_proc(p, &kp);
 			sx_sunlock(&proctree_lock);
 			if (p == NULL || kp.ki_stat == SIDL) {
 #ifdef illumos
 				mutex_exit(&pidlock);
 #endif
 				return (ESRCH);
 			}
 #ifdef illumos
 			mutex_enter(&p->p_lock);
 			mutex_exit(&pidlock);
 #else
 			PROC_LOCK_ASSERT(p, MA_OWNED);
 #endif
 
 #ifdef notyet
 			if ((ret = priv_proc_cred_perm(cr, p, NULL,
 			    VREAD)) != 0) {
 #ifdef illumos
 				mutex_exit(&p->p_lock);
 #else
 				PROC_UNLOCK(p);
 #endif
 				return (ret);
 			}
 #endif /* notyet */
 
 #ifdef illumos
 			mutex_exit(&p->p_lock);
 #else
 			PROC_UNLOCK(p);
 #endif
 		}
 #endif /* notyet */
 
 		index = FASTTRAP_TPOINTS_INDEX(instr.ftiq_pid, instr.ftiq_pc);
 
 		mutex_enter(&fasttrap_tpoints.fth_table[index].ftb_mtx);
 		tp = fasttrap_tpoints.fth_table[index].ftb_data;
 		while (tp != NULL) {
 			if (instr.ftiq_pid == tp->ftt_pid &&
 			    instr.ftiq_pc == tp->ftt_pc &&
 			    tp->ftt_proc->ftpc_acount != 0)
 				break;
 
 			tp = tp->ftt_next;
 		}
 
 		if (tp == NULL) {
 			mutex_exit(&fasttrap_tpoints.fth_table[index].ftb_mtx);
 			return (ENOENT);
 		}
 
 		bcopy(&tp->ftt_instr, &instr.ftiq_instr,
 		    sizeof (instr.ftiq_instr));
 		mutex_exit(&fasttrap_tpoints.fth_table[index].ftb_mtx);
 
 		if (copyout(&instr, (void *)arg, sizeof (instr)) != 0)
 			return (EFAULT);
 
 		return (0);
 	}
 
 	return (EINVAL);
 }
 
 static int
 fasttrap_load(void)
 {
 	ulong_t nent;
 	int i, ret;
 
         /* Create the /dev/dtrace/fasttrap entry. */
         fasttrap_cdev = make_dev(&fasttrap_cdevsw, 0, UID_ROOT, GID_WHEEL, 0600,
             "dtrace/fasttrap");
 
 	mtx_init(&fasttrap_cleanup_mtx, "fasttrap clean", "dtrace", MTX_DEF);
 	mutex_init(&fasttrap_count_mtx, "fasttrap count mtx", MUTEX_DEFAULT,
 	    NULL);
 
 #ifdef illumos
 	fasttrap_max = ddi_getprop(DDI_DEV_T_ANY, devi, DDI_PROP_DONTPASS,
 	    "fasttrap-max-probes", FASTTRAP_MAX_DEFAULT);
 #else
 	fasttrap_max = FASTTRAP_MAX_DEFAULT;
 #endif
 	fasttrap_total = 0;
 
 	/*
 	 * Conjure up the tracepoints hashtable...
 	 */
 #ifdef illumos
 	nent = ddi_getprop(DDI_DEV_T_ANY, devi, DDI_PROP_DONTPASS,
 	    "fasttrap-hash-size", FASTTRAP_TPOINTS_DEFAULT_SIZE);
 #else
 	nent = FASTTRAP_TPOINTS_DEFAULT_SIZE;
 #endif
 
 	if (nent == 0 || nent > 0x1000000)
 		nent = FASTTRAP_TPOINTS_DEFAULT_SIZE;
 
 	if (ISP2(nent))
 		fasttrap_tpoints.fth_nent = nent;
 	else
 		fasttrap_tpoints.fth_nent = 1 << fasttrap_highbit(nent);
 	ASSERT(fasttrap_tpoints.fth_nent > 0);
 	fasttrap_tpoints.fth_mask = fasttrap_tpoints.fth_nent - 1;
 	fasttrap_tpoints.fth_table = kmem_zalloc(fasttrap_tpoints.fth_nent *
 	    sizeof (fasttrap_bucket_t), KM_SLEEP);
 #ifndef illumos
 	for (i = 0; i < fasttrap_tpoints.fth_nent; i++)
 		mutex_init(&fasttrap_tpoints.fth_table[i].ftb_mtx,
 		    "tracepoints bucket mtx", MUTEX_DEFAULT, NULL);
 #endif
 
 	/*
 	 * ... and the providers hash table...
 	 */
 	nent = FASTTRAP_PROVIDERS_DEFAULT_SIZE;
 	if (ISP2(nent))
 		fasttrap_provs.fth_nent = nent;
 	else
 		fasttrap_provs.fth_nent = 1 << fasttrap_highbit(nent);
 	ASSERT(fasttrap_provs.fth_nent > 0);
 	fasttrap_provs.fth_mask = fasttrap_provs.fth_nent - 1;
 	fasttrap_provs.fth_table = kmem_zalloc(fasttrap_provs.fth_nent *
 	    sizeof (fasttrap_bucket_t), KM_SLEEP);
 #ifndef illumos
 	for (i = 0; i < fasttrap_provs.fth_nent; i++)
 		mutex_init(&fasttrap_provs.fth_table[i].ftb_mtx, 
 		    "providers bucket mtx", MUTEX_DEFAULT, NULL);
 #endif
 
 	ret = kproc_create(fasttrap_pid_cleanup_cb, NULL,
 	    &fasttrap_cleanup_proc, 0, 0, "ftcleanup");
 	if (ret != 0) {
 		destroy_dev(fasttrap_cdev);
 #ifndef illumos
 		for (i = 0; i < fasttrap_provs.fth_nent; i++)
 			mutex_destroy(&fasttrap_provs.fth_table[i].ftb_mtx);
 		for (i = 0; i < fasttrap_tpoints.fth_nent; i++)
 			mutex_destroy(&fasttrap_tpoints.fth_table[i].ftb_mtx);
 #endif
 		kmem_free(fasttrap_provs.fth_table, fasttrap_provs.fth_nent *
 		    sizeof (fasttrap_bucket_t));
 		mtx_destroy(&fasttrap_cleanup_mtx);
 		mutex_destroy(&fasttrap_count_mtx);
 		return (ret);
 	}
 
 
 	/*
 	 * ... and the procs hash table.
 	 */
 	nent = FASTTRAP_PROCS_DEFAULT_SIZE;
 	if (ISP2(nent))
 		fasttrap_procs.fth_nent = nent;
 	else
 		fasttrap_procs.fth_nent = 1 << fasttrap_highbit(nent);
 	ASSERT(fasttrap_procs.fth_nent > 0);
 	fasttrap_procs.fth_mask = fasttrap_procs.fth_nent - 1;
 	fasttrap_procs.fth_table = kmem_zalloc(fasttrap_procs.fth_nent *
 	    sizeof (fasttrap_bucket_t), KM_SLEEP);
 #ifndef illumos
 	for (i = 0; i < fasttrap_procs.fth_nent; i++)
 		mutex_init(&fasttrap_procs.fth_table[i].ftb_mtx,
 		    "processes bucket mtx", MUTEX_DEFAULT, NULL);
 
 	CPU_FOREACH(i) {
 		mutex_init(&fasttrap_cpuc_pid_lock[i], "fasttrap barrier",
 		    MUTEX_DEFAULT, NULL);
 	}
 
 	/*
 	 * This event handler must run before kdtrace_thread_dtor() since it
 	 * accesses the thread's struct kdtrace_thread.
 	 */
 	fasttrap_thread_dtor_tag = EVENTHANDLER_REGISTER(thread_dtor,
 	    fasttrap_thread_dtor, NULL, EVENTHANDLER_PRI_FIRST);
 #endif
 
 	/*
 	 * Install our hooks into fork(2), exec(2), and exit(2).
 	 */
 	dtrace_fasttrap_fork = &fasttrap_fork;
 	dtrace_fasttrap_exit = &fasttrap_exec_exit;
 	dtrace_fasttrap_exec = &fasttrap_exec_exit;
 
 	(void) dtrace_meta_register("fasttrap", &fasttrap_mops, NULL,
 	    &fasttrap_meta_id);
 
 	return (0);
 }
 
 static int
 fasttrap_unload(void)
 {
 	int i, fail = 0;
 
 	/*
 	 * Unregister the meta-provider to make sure no new fasttrap-
 	 * managed providers come along while we're trying to close up
 	 * shop. If we fail to detach, we'll need to re-register as a
 	 * meta-provider. We can fail to unregister as a meta-provider
 	 * if providers we manage still exist.
 	 */
 	if (fasttrap_meta_id != DTRACE_METAPROVNONE &&
 	    dtrace_meta_unregister(fasttrap_meta_id) != 0)
 		return (-1);
 
 	/*
 	 * Iterate over all of our providers. If there's still a process
 	 * that corresponds to that pid, fail to detach.
 	 */
 	for (i = 0; i < fasttrap_provs.fth_nent; i++) {
 		fasttrap_provider_t **fpp, *fp;
 		fasttrap_bucket_t *bucket = &fasttrap_provs.fth_table[i];
 
 		mutex_enter(&bucket->ftb_mtx);
 		fpp = (fasttrap_provider_t **)&bucket->ftb_data;
 		while ((fp = *fpp) != NULL) {
 			/*
 			 * Acquire and release the lock as a simple way of
 			 * waiting for any other consumer to finish with
 			 * this provider. A thread must first acquire the
 			 * bucket lock so there's no chance of another thread
 			 * blocking on the provider's lock.
 			 */
 			mutex_enter(&fp->ftp_mtx);
 			mutex_exit(&fp->ftp_mtx);
 
 			if (dtrace_unregister(fp->ftp_provid) != 0) {
 				fail = 1;
 				fpp = &fp->ftp_next;
 			} else {
 				*fpp = fp->ftp_next;
 				fasttrap_provider_free(fp);
 			}
 		}
 
 		mutex_exit(&bucket->ftb_mtx);
 	}
 
 	if (fail) {
 		(void) dtrace_meta_register("fasttrap", &fasttrap_mops, NULL,
 		    &fasttrap_meta_id);
 
 		return (-1);
 	}
 
 	/*
 	 * Stop new processes from entering these hooks now, before the
 	 * fasttrap_cleanup thread runs.  That way all processes will hopefully
 	 * be out of these hooks before we free fasttrap_provs.fth_table
 	 */
 	ASSERT(dtrace_fasttrap_fork == &fasttrap_fork);
 	dtrace_fasttrap_fork = NULL;
 
 	ASSERT(dtrace_fasttrap_exec == &fasttrap_exec_exit);
 	dtrace_fasttrap_exec = NULL;
 
 	ASSERT(dtrace_fasttrap_exit == &fasttrap_exec_exit);
 	dtrace_fasttrap_exit = NULL;
 
 	mtx_lock(&fasttrap_cleanup_mtx);
 	fasttrap_cleanup_drain = 1;
 	/* Wait for the cleanup thread to finish up and signal us. */
 	wakeup(&fasttrap_cleanup_cv);
 	mtx_sleep(&fasttrap_cleanup_drain, &fasttrap_cleanup_mtx, 0, "ftcld",
 	    0);
 	fasttrap_cleanup_proc = NULL;
 	mtx_destroy(&fasttrap_cleanup_mtx);
 
 #ifdef DEBUG
 	mutex_enter(&fasttrap_count_mtx);
 	ASSERT(fasttrap_pid_count == 0);
 	mutex_exit(&fasttrap_count_mtx);
 #endif
 
 #ifndef illumos
 	EVENTHANDLER_DEREGISTER(thread_dtor, fasttrap_thread_dtor_tag);
 
 	for (i = 0; i < fasttrap_tpoints.fth_nent; i++)
 		mutex_destroy(&fasttrap_tpoints.fth_table[i].ftb_mtx);
 	for (i = 0; i < fasttrap_provs.fth_nent; i++)
 		mutex_destroy(&fasttrap_provs.fth_table[i].ftb_mtx);
 	for (i = 0; i < fasttrap_procs.fth_nent; i++)
 		mutex_destroy(&fasttrap_procs.fth_table[i].ftb_mtx);
 #endif
 	kmem_free(fasttrap_tpoints.fth_table,
 	    fasttrap_tpoints.fth_nent * sizeof (fasttrap_bucket_t));
 	fasttrap_tpoints.fth_nent = 0;
 
 	kmem_free(fasttrap_provs.fth_table,
 	    fasttrap_provs.fth_nent * sizeof (fasttrap_bucket_t));
 	fasttrap_provs.fth_nent = 0;
 
 	kmem_free(fasttrap_procs.fth_table,
 	    fasttrap_procs.fth_nent * sizeof (fasttrap_bucket_t));
 	fasttrap_procs.fth_nent = 0;
 
 #ifndef illumos
 	destroy_dev(fasttrap_cdev);
 	mutex_destroy(&fasttrap_count_mtx);
 	CPU_FOREACH(i) {
 		mutex_destroy(&fasttrap_cpuc_pid_lock[i]);
 	}
 #endif
 
 	return (0);
 }
 
 /* ARGSUSED */
 static int
 fasttrap_modevent(module_t mod __unused, int type, void *data __unused)
 {
 	int error = 0;
 
 	switch (type) {
 	case MOD_LOAD:
 		break;
 
 	case MOD_UNLOAD:
 		break;
 
 	case MOD_SHUTDOWN:
 		break;
 
 	default:
 		error = EOPNOTSUPP;
 		break;
 	}
 	return (error);
 }
 
 SYSINIT(fasttrap_load, SI_SUB_DTRACE_PROVIDER, SI_ORDER_ANY, fasttrap_load,
     NULL);
 SYSUNINIT(fasttrap_unload, SI_SUB_DTRACE_PROVIDER, SI_ORDER_ANY,
     fasttrap_unload, NULL);
 
 DEV_MODULE(fasttrap, fasttrap_modevent, NULL);
 MODULE_VERSION(fasttrap, 1);
 MODULE_DEPEND(fasttrap, dtrace, 1, 1, 1);
 MODULE_DEPEND(fasttrap, opensolaris, 1, 1, 1);
Index: projects/clang360-import/sys/cddl/contrib/opensolaris
===================================================================
--- projects/clang360-import/sys/cddl/contrib/opensolaris	(revision 277944)
+++ projects/clang360-import/sys/cddl/contrib/opensolaris	(revision 277945)

Property changes on: projects/clang360-import/sys/cddl/contrib/opensolaris
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/sys/cddl/contrib/opensolaris:r277844-277944
Index: projects/clang360-import/sys/conf/files.amd64
===================================================================
--- projects/clang360-import/sys/conf/files.amd64	(revision 277944)
+++ projects/clang360-import/sys/conf/files.amd64	(revision 277945)
@@ -1,582 +1,582 @@
 # This file tells config what files go into building a kernel,
 # files marked standard are always included.
 #
 # $FreeBSD$
 #
 # The long compile-with and dependency lines are required because of
 # limitations in config: backslash-newline doesn't work in strings, and
 # dependency lines other than the first are silently ignored.
 #
 #
 linux32_genassym.o		optional	compat_linux32		\
 	dependency 	"$S/amd64/linux32/linux32_genassym.c"		\
 	compile-with	"${CC} ${CFLAGS:N-fno-common} -c ${.IMPSRC}"	\
 	no-obj no-implicit-rule						\
 	clean		"linux32_genassym.o"
 #
 linux32_assym.h			optional	compat_linux32		\
 	dependency 	"$S/kern/genassym.sh linux32_genassym.o"	\
 	compile-with	"sh $S/kern/genassym.sh linux32_genassym.o > ${.TARGET}" \
 	no-obj no-implicit-rule before-depend				\
 	clean		"linux32_assym.h"
 #
 ia32_genassym.o			standard				\
 	dependency 	"$S/compat/ia32/ia32_genassym.c"		\
 	compile-with	"${CC} ${CFLAGS:N-fno-common} -c ${.IMPSRC}"	\
 	no-obj no-implicit-rule						\
 	clean		"ia32_genassym.o"
 #
 ia32_assym.h			standard				\
 	dependency 	"$S/kern/genassym.sh ia32_genassym.o"		\
 	compile-with	"env NM='${NM}' sh $S/kern/genassym.sh ia32_genassym.o > ${.TARGET}" \
 	no-obj no-implicit-rule before-depend				\
 	clean		"ia32_assym.h"
 #
 font.h				optional	sc_dflt_font		\
 	compile-with	"uudecode < /usr/share/syscons/fonts/${SC_DFLT_FONT}-8x16.fnt && file2c 'static u_char dflt_font_16[16*256] = {' '};' < ${SC_DFLT_FONT}-8x16 > font.h && uudecode < /usr/share/syscons/fonts/${SC_DFLT_FONT}-8x14.fnt && file2c 'static u_char dflt_font_14[14*256] = {' '};' < ${SC_DFLT_FONT}-8x14 >> font.h && uudecode < /usr/share/syscons/fonts/${SC_DFLT_FONT}-8x8.fnt && file2c 'static u_char dflt_font_8[8*256] = {' '};' < ${SC_DFLT_FONT}-8x8 >> font.h"									\
 	no-obj no-implicit-rule before-depend				\
 	clean		"font.h ${SC_DFLT_FONT}-8x14 ${SC_DFLT_FONT}-8x16 ${SC_DFLT_FONT}-8x8"
 #
 atkbdmap.h			optional	atkbd_dflt_keymap	\
 	compile-with	"/usr/sbin/kbdcontrol -L ${ATKBD_DFLT_KEYMAP} | sed -e 's/^static keymap_t.* = /static keymap_t key_map = /' -e 's/^static accentmap_t.* = /static accentmap_t accent_map = /' > atkbdmap.h"			\
 	no-obj no-implicit-rule before-depend				\
 	clean		"atkbdmap.h"
 #
 ukbdmap.h			optional	ukbd_dflt_keymap	\
 	compile-with	"/usr/sbin/kbdcontrol -L ${UKBD_DFLT_KEYMAP} | sed -e 's/^static keymap_t.* = /static keymap_t key_map = /' -e 's/^static accentmap_t.* = /static accentmap_t accent_map = /' > ukbdmap.h"			\
 	no-obj no-implicit-rule before-depend				\
 	clean		"ukbdmap.h"
 #
 hpt27xx_lib.o			optional	hpt27xx			\
 	dependency	"$S/dev/hpt27xx/amd64-elf.hpt27xx_lib.o.uu"	\
 	compile-with	"uudecode < $S/dev/hpt27xx/amd64-elf.hpt27xx_lib.o.uu" \
 	no-implicit-rule
 #
 hptmvraid.o			optional	hptmv			\
 	dependency	"$S/dev/hptmv/amd64-elf.raid.o.uu"	\
 	compile-with	"uudecode < $S/dev/hptmv/amd64-elf.raid.o.uu" \
 	no-implicit-rule
 #
 hptnr_lib.o			optional	hptnr			\
 	dependency	"$S/dev/hptnr/amd64-elf.hptnr_lib.o.uu"	\
 	compile-with	"uudecode < $S/dev/hptnr/amd64-elf.hptnr_lib.o.uu" \
 	no-implicit-rule
 #
 hptrr_lib.o			optional	hptrr			\
 	dependency	"$S/dev/hptrr/amd64-elf.hptrr_lib.o.uu"		\
 	compile-with	"uudecode < $S/dev/hptrr/amd64-elf.hptrr_lib.o.uu" \
 	no-implicit-rule
 #
 amd64/acpica/acpi_machdep.c	optional	acpi
 acpi_wakecode.o			optional	acpi			\
 	dependency	"$S/amd64/acpica/acpi_wakecode.S assym.s"	\
 	compile-with	"${NORMAL_S}"					\
 	no-obj no-implicit-rule before-depend				\
 	clean		"acpi_wakecode.o"
 acpi_wakecode.bin		optional	acpi			\
 	dependency	"acpi_wakecode.o"				\
 	compile-with	"${OBJCOPY} -S -O binary acpi_wakecode.o ${.TARGET}" \
 	no-obj no-implicit-rule	before-depend				\
 	clean		"acpi_wakecode.bin"
 acpi_wakecode.h			optional	acpi			\
 	dependency	"acpi_wakecode.bin"				\
 	compile-with	"file2c -sx 'static char wakecode[] = {' '};' < acpi_wakecode.bin > ${.TARGET}" \
 	no-obj no-implicit-rule	before-depend				\
 	clean		"acpi_wakecode.h"
 acpi_wakedata.h			optional	acpi			\
 	dependency	"acpi_wakecode.o"				\
 	compile-with	'${NM} -n --defined-only acpi_wakecode.o | while read offset dummy what; do echo "#define	$${what}	0x$${offset}"; done > ${.TARGET}' \
 	no-obj no-implicit-rule	before-depend				\
 	clean		"acpi_wakedata.h"
 #
 amd64/amd64/amd64_mem.c		optional	mem
 #amd64/amd64/apic_vector.S	standard
 amd64/amd64/atomic.c		standard
 amd64/amd64/autoconf.c		standard
 amd64/amd64/bios.c		standard
 amd64/amd64/bpf_jit_machdep.c	optional	bpf_jitter
 amd64/amd64/cpu_switch.S	standard
 amd64/amd64/db_disasm.c		optional	ddb
 amd64/amd64/db_interface.c	optional	ddb
 amd64/amd64/db_trace.c		optional	ddb
 amd64/amd64/elf_machdep.c	standard
 amd64/amd64/exception.S		standard
 amd64/amd64/fpu.c		standard
 amd64/amd64/gdb_machdep.c	optional	gdb
 amd64/amd64/in_cksum.c		optional	inet | inet6
 amd64/amd64/initcpu.c		standard
 amd64/amd64/io.c		optional	io
 amd64/amd64/locore.S		standard	no-obj
 amd64/amd64/xen-locore.S	optional	xenhvm
 amd64/amd64/machdep.c		standard
 amd64/amd64/mem.c		optional	mem
 amd64/amd64/minidump_machdep.c	standard
 amd64/amd64/mp_machdep.c	optional	smp
 amd64/amd64/mp_watchdog.c	optional	mp_watchdog smp
 amd64/amd64/mpboot.S		optional	smp
 amd64/amd64/pmap.c		standard
 amd64/amd64/prof_machdep.c	optional	profiling-routine
 amd64/amd64/ptrace_machdep.c	standard
 amd64/amd64/sigtramp.S		standard
 amd64/amd64/stack_machdep.c	optional	ddb | stack
 amd64/amd64/support.S		standard
 amd64/amd64/sys_machdep.c	standard
 amd64/amd64/trap.c		standard
 amd64/amd64/uio_machdep.c	standard
 amd64/amd64/uma_machdep.c	standard
 amd64/amd64/vm_machdep.c	standard
 amd64/pci/pci_cfgreg.c		optional	pci
 cddl/contrib/opensolaris/common/atomic/amd64/opensolaris_atomic.S	optional zfs compile-with "${ZFS_S}"
 crypto/aesni/aeskeys_amd64.S	optional aesni
 crypto/aesni/aesni.c		optional aesni
 aesni_ghash.o			optional aesni				\
 	dependency	"$S/crypto/aesni/aesni_ghash.c"			\
-	compile-with	"${CC} -c ${CFLAGS:C/^-O2$/-O3/:N-nostdinc} ${WERROR} ${PROF} -mmmx -msse -msse4 -maes -mpclmul ${.IMPSRC}" \
+	compile-with	"${CC} -c ${CFLAGS:C/^-O2$/-O3/:N-nostdinc} ${WERROR} ${NO_WCAST_QUAL} ${PROF} -mmmx -msse -msse4 -maes -mpclmul ${.IMPSRC}" \
 	no-implicit-rule						\
 	clean		"aesni_ghash.o"
 aesni_wrap.o			optional aesni				\
 	dependency	"$S/crypto/aesni/aesni_wrap.c"			\
-	compile-with	"${CC} -c ${CFLAGS:C/^-O2$/-O3/:N-nostdinc} ${WERROR} ${PROF} -mmmx -msse -msse4 -maes ${.IMPSRC}" \
+	compile-with	"${CC} -c ${CFLAGS:C/^-O2$/-O3/:N-nostdinc} ${WERROR} ${NO_WCAST_QUAL} ${PROF} -mmmx -msse -msse4 -maes ${.IMPSRC}" \
 	no-implicit-rule						\
 	clean		"aesni_wrap.o"
 crypto/blowfish/bf_enc.c	optional	crypto | ipsec
 crypto/des/des_enc.c		optional	crypto | ipsec | netsmb
 crypto/via/padlock.c		optional	padlock
 crypto/via/padlock_cipher.c	optional	padlock
 crypto/via/padlock_hash.c	optional	padlock
 dev/acpica/acpi_if.m		standard
 dev/acpi_support/acpi_wmi_if.m	standard
 dev/agp/agp_amd64.c		optional	agp
 dev/agp/agp_i810.c		optional	agp
 dev/agp/agp_via.c		optional	agp
 dev/amdsbwd/amdsbwd.c		optional	amdsbwd
 dev/amdtemp/amdtemp.c		optional	amdtemp
 dev/arcmsr/arcmsr.c		optional	arcmsr pci
 dev/asmc/asmc.c			optional	asmc isa
 dev/atkbdc/atkbd.c		optional	atkbd atkbdc
 dev/atkbdc/atkbd_atkbdc.c	optional	atkbd atkbdc
 dev/atkbdc/atkbdc.c		optional	atkbdc
 dev/atkbdc/atkbdc_isa.c		optional	atkbdc isa
 dev/atkbdc/atkbdc_subr.c	optional	atkbdc
 dev/atkbdc/psm.c		optional	psm atkbdc
 dev/bxe/bxe.c                   optional	bxe pci
 dev/bxe/bxe_stats.c             optional	bxe pci
 dev/bxe/bxe_debug.c             optional	bxe pci
 dev/bxe/ecore_sp.c              optional	bxe pci
 dev/bxe/bxe_elink.c             optional	bxe pci
 dev/bxe/57710_init_values.c     optional	bxe pci
 dev/bxe/57711_init_values.c     optional	bxe pci
 dev/bxe/57712_init_values.c     optional	bxe pci
 dev/coretemp/coretemp.c		optional	coretemp
 dev/cpuctl/cpuctl.c		optional	cpuctl
 dev/dpms/dpms.c			optional	dpms
 # There are no systems with isa slots, so all ed isa entries should go..
 dev/ed/if_ed_3c503.c		optional	ed isa ed_3c503
 dev/ed/if_ed_isa.c		optional	ed isa
 dev/ed/if_ed_wd80x3.c		optional	ed isa
 dev/ed/if_ed_hpp.c		optional	ed isa ed_hpp
 dev/ed/if_ed_sic.c		optional	ed isa ed_sic
 dev/fb/fb.c			optional	fb | vga
 dev/fb/s3_pci.c			optional	s3pci
 dev/fb/vesa.c			optional	vga vesa
 dev/fb/vga.c			optional	vga
 dev/ichwd/ichwd.c		optional	ichwd
 dev/if_ndis/if_ndis.c		optional	ndis
 dev/if_ndis/if_ndis_pccard.c	optional	ndis pccard
 dev/if_ndis/if_ndis_pci.c	optional	ndis cardbus | ndis pci
 dev/if_ndis/if_ndis_usb.c	optional	ndis usb
 dev/io/iodev.c			optional	io
 dev/ipmi/ipmi.c			optional	ipmi
 dev/ipmi/ipmi_acpi.c		optional	ipmi acpi
 dev/ipmi/ipmi_isa.c		optional	ipmi isa
 dev/ipmi/ipmi_kcs.c		optional	ipmi
 dev/ipmi/ipmi_smic.c		optional	ipmi
 dev/ipmi/ipmi_smbus.c		optional	ipmi smbus
 dev/ipmi/ipmi_smbios.c		optional	ipmi
 dev/ipmi/ipmi_ssif.c		optional	ipmi smbus
 dev/ipmi/ipmi_pci.c		optional	ipmi pci
 dev/ipmi/ipmi_linux.c		optional	ipmi compat_linux32
 dev/ixl/if_ixl.c		optional	ixl pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/if_ixlv.c		optional	ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/ixlvc.c			optional	ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/ixl_txrx.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/i40e_osdep.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/i40e_lan_hmc.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/i40e_hmc.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/i40e_common.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/i40e_nvm.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/ixl/i40e_adminq.c		optional	ixl pci | ixlv pci \
 	compile-with "${NORMAL_C} -I$S/dev/ixl"
 dev/fdc/fdc.c			optional	fdc
 dev/fdc/fdc_acpi.c		optional	fdc
 dev/fdc/fdc_isa.c		optional	fdc isa
 dev/fdc/fdc_pccard.c		optional	fdc pccard
 dev/fdt/fdt_x86.c		optional	fdt
 dev/hpt27xx/hpt27xx_os_bsd.c	optional	hpt27xx
 dev/hpt27xx/hpt27xx_osm_bsd.c	optional	hpt27xx
 dev/hpt27xx/hpt27xx_config.c	optional	hpt27xx
 dev/hptmv/entry.c		optional	hptmv
 dev/hptmv/mv.c			optional	hptmv
 dev/hptmv/gui_lib.c		optional	hptmv
 dev/hptmv/hptproc.c		optional	hptmv
 dev/hptmv/ioctl.c		optional	hptmv
 dev/hptnr/hptnr_os_bsd.c	optional	hptnr
 dev/hptnr/hptnr_osm_bsd.c	optional	hptnr
 dev/hptnr/hptnr_config.c	optional	hptnr
 dev/hptrr/hptrr_os_bsd.c	optional	hptrr
 dev/hptrr/hptrr_osm_bsd.c	optional	hptrr
 dev/hptrr/hptrr_config.c	optional	hptrr
 dev/hwpmc/hwpmc_amd.c		optional	hwpmc
 dev/hwpmc/hwpmc_intel.c		optional	hwpmc
 dev/hwpmc/hwpmc_core.c		optional	hwpmc
 dev/hwpmc/hwpmc_uncore.c	optional	hwpmc
 dev/hwpmc/hwpmc_piv.c		optional	hwpmc
 dev/hwpmc/hwpmc_tsc.c		optional	hwpmc
 dev/hwpmc/hwpmc_x86.c		optional	hwpmc
 dev/hyperv/netvsc/hv_net_vsc.c				optional	hyperv
 dev/hyperv/netvsc/hv_netvsc_drv_freebsd.c		optional	hyperv
 dev/hyperv/netvsc/hv_rndis_filter.c			optional	hyperv
 dev/hyperv/stordisengage/hv_ata_pci_disengage.c		optional	hyperv
 dev/hyperv/storvsc/hv_storvsc_drv_freebsd.c		optional	hyperv
 dev/hyperv/utilities/hv_kvp.c				optional	hyperv
 dev/hyperv/utilities/hv_util.c				optional	hyperv
 dev/hyperv/vmbus/hv_channel.c				optional	hyperv
 dev/hyperv/vmbus/hv_channel_mgmt.c			optional	hyperv
 dev/hyperv/vmbus/hv_connection.c			optional	hyperv
 dev/hyperv/vmbus/hv_hv.c				optional	hyperv
 dev/hyperv/vmbus/hv_ring_buffer.c			optional	hyperv
 dev/hyperv/vmbus/hv_vmbus_drv_freebsd.c			optional	hyperv
 dev/kbd/kbd.c			optional	atkbd | sc | ukbd | vt
 dev/nfe/if_nfe.c		optional	nfe pci
 dev/ntb/if_ntb/if_ntb.c		optional	if_ntb
 dev/ntb/ntb_hw/ntb_hw.c		optional	if_ntb ntb_hw
 dev/nvd/nvd.c			optional	nvd nvme
 dev/nvme/nvme.c			optional	nvme
 dev/nvme/nvme_ctrlr.c		optional	nvme
 dev/nvme/nvme_ctrlr_cmd.c	optional	nvme
 dev/nvme/nvme_ns.c		optional	nvme
 dev/nvme/nvme_ns_cmd.c		optional	nvme
 dev/nvme/nvme_qpair.c		optional	nvme
 dev/nvme/nvme_sysctl.c		optional	nvme
 dev/nvme/nvme_test.c		optional	nvme
 dev/nvme/nvme_util.c		optional	nvme
 dev/nvram/nvram.c		optional	nvram isa
 dev/random/ivy.c		optional	rdrand_rng
 dev/random/nehemiah.c		optional	padlock_rng
 dev/qlxge/qls_dbg.c		optional	qlxge pci
 dev/qlxge/qls_dump.c		optional	qlxge pci
 dev/qlxge/qls_hw.c		optional	qlxge pci
 dev/qlxge/qls_ioctl.c		optional	qlxge pci
 dev/qlxge/qls_isr.c		optional	qlxge pci
 dev/qlxge/qls_os.c		optional	qlxge pci
 dev/qlxgb/qla_dbg.c		optional	qlxgb pci
 dev/qlxgb/qla_hw.c		optional	qlxgb pci
 dev/qlxgb/qla_ioctl.c		optional	qlxgb pci
 dev/qlxgb/qla_isr.c		optional	qlxgb pci
 dev/qlxgb/qla_misc.c		optional	qlxgb pci
 dev/qlxgb/qla_os.c		optional	qlxgb pci
 dev/qlxgbe/ql_dbg.c		optional	qlxgbe pci
 dev/qlxgbe/ql_hw.c		optional	qlxgbe pci
 dev/qlxgbe/ql_ioctl.c		optional	qlxgbe pci
 dev/qlxgbe/ql_isr.c		optional	qlxgbe pci
 dev/qlxgbe/ql_misc.c		optional	qlxgbe pci
 dev/qlxgbe/ql_os.c		optional	qlxgbe pci
 dev/qlxgbe/ql_reset.c		optional	qlxgbe pci
 dev/sfxge/common/efx_bootcfg.c	optional sfxge inet pci
 dev/sfxge/common/efx_ev.c	optional sfxge inet pci
 dev/sfxge/common/efx_filter.c	optional sfxge inet pci
 dev/sfxge/common/efx_intr.c	optional sfxge inet pci
 dev/sfxge/common/efx_mac.c	optional sfxge inet pci
 dev/sfxge/common/efx_mcdi.c	optional sfxge inet pci
 dev/sfxge/common/efx_mon.c	optional sfxge inet pci
 dev/sfxge/common/efx_nic.c	optional sfxge inet pci
 dev/sfxge/common/efx_nvram.c	optional sfxge inet pci
 dev/sfxge/common/efx_phy.c	optional sfxge inet pci
 dev/sfxge/common/efx_port.c	optional sfxge inet pci
 dev/sfxge/common/efx_rx.c	optional sfxge inet pci
 dev/sfxge/common/efx_sram.c	optional sfxge inet pci
 dev/sfxge/common/efx_tx.c	optional sfxge inet pci
 dev/sfxge/common/efx_vpd.c	optional sfxge inet pci
 dev/sfxge/common/efx_wol.c	optional sfxge inet pci
 dev/sfxge/common/siena_mac.c	optional sfxge inet pci
 dev/sfxge/common/siena_mon.c	optional sfxge inet pci
 dev/sfxge/common/siena_nic.c	optional sfxge inet pci
 dev/sfxge/common/siena_nvram.c	optional sfxge inet pci
 dev/sfxge/common/siena_phy.c	optional sfxge inet pci
 dev/sfxge/common/siena_sram.c	optional sfxge inet pci
 dev/sfxge/common/siena_vpd.c	optional sfxge inet pci
 dev/sfxge/sfxge.c		optional sfxge inet pci
 dev/sfxge/sfxge_dma.c		optional sfxge inet pci
 dev/sfxge/sfxge_ev.c		optional sfxge inet pci
 dev/sfxge/sfxge_intr.c		optional sfxge inet pci
 dev/sfxge/sfxge_mcdi.c		optional sfxge inet pci
 dev/sfxge/sfxge_port.c		optional sfxge inet pci
 dev/sfxge/sfxge_rx.c		optional sfxge inet pci
 dev/sfxge/sfxge_tx.c		optional sfxge inet pci
 dev/sio/sio.c			optional	sio
 dev/sio/sio_isa.c		optional	sio isa
 dev/sio/sio_pccard.c		optional	sio pccard
 dev/sio/sio_pci.c		optional	sio pci
 dev/sio/sio_puc.c		optional	sio puc
 dev/speaker/spkr.c		optional	speaker
 dev/syscons/apm/apm_saver.c	optional	apm_saver apm
 dev/syscons/scterm-teken.c	optional	sc
 dev/syscons/scvesactl.c		optional	sc vga vesa
 dev/syscons/scvgarndr.c		optional	sc vga
 dev/syscons/scvtb.c		optional	sc
 dev/tpm/tpm.c			optional	tpm
 dev/tpm/tpm_acpi.c		optional	tpm acpi
 dev/tpm/tpm_isa.c		optional	tpm isa
 dev/uart/uart_cpu_x86.c		optional	uart
 dev/viawd/viawd.c		optional	viawd
 dev/vmware/vmxnet3/if_vmx.c	optional	vmx
 dev/wbwd/wbwd.c			optional	wbwd
 dev/wpi/if_wpi.c		optional	wpi
 dev/xen/pci/xen_acpi_pci.c	optional	xenhvm
 dev/xen/pci/xen_pci.c		optional	xenhvm
 dev/isci/isci.c							optional isci
 dev/isci/isci_controller.c					optional isci
 dev/isci/isci_domain.c						optional isci
 dev/isci/isci_interrupt.c					optional isci
 dev/isci/isci_io_request.c					optional isci
 dev/isci/isci_logger.c						optional isci
 dev/isci/isci_oem_parameters.c					optional isci
 dev/isci/isci_remote_device.c					optional isci
 dev/isci/isci_sysctl.c						optional isci
 dev/isci/isci_task_request.c					optional isci
 dev/isci/isci_timer.c						optional isci
 dev/isci/scil/sati.c						optional isci
 dev/isci/scil/sati_abort_task_set.c				optional isci
 dev/isci/scil/sati_atapi.c					optional isci
 dev/isci/scil/sati_device.c					optional isci
 dev/isci/scil/sati_inquiry.c					optional isci
 dev/isci/scil/sati_log_sense.c					optional isci
 dev/isci/scil/sati_lun_reset.c					optional isci
 dev/isci/scil/sati_mode_pages.c					optional isci
 dev/isci/scil/sati_mode_select.c				optional isci
 dev/isci/scil/sati_mode_sense.c					optional isci
 dev/isci/scil/sati_mode_sense_10.c				optional isci
 dev/isci/scil/sati_mode_sense_6.c				optional isci
 dev/isci/scil/sati_move.c					optional isci
 dev/isci/scil/sati_passthrough.c				optional isci
 dev/isci/scil/sati_read.c					optional isci
 dev/isci/scil/sati_read_buffer.c				optional isci
 dev/isci/scil/sati_read_capacity.c				optional isci
 dev/isci/scil/sati_reassign_blocks.c				optional isci
 dev/isci/scil/sati_report_luns.c				optional isci
 dev/isci/scil/sati_request_sense.c				optional isci
 dev/isci/scil/sati_start_stop_unit.c				optional isci
 dev/isci/scil/sati_synchronize_cache.c				optional isci
 dev/isci/scil/sati_test_unit_ready.c				optional isci
 dev/isci/scil/sati_unmap.c					optional isci
 dev/isci/scil/sati_util.c					optional isci
 dev/isci/scil/sati_verify.c					optional isci
 dev/isci/scil/sati_write.c					optional isci
 dev/isci/scil/sati_write_and_verify.c				optional isci
 dev/isci/scil/sati_write_buffer.c				optional isci
 dev/isci/scil/sati_write_long.c					optional isci
 dev/isci/scil/sci_abstract_list.c				optional isci
 dev/isci/scil/sci_base_controller.c				optional isci
 dev/isci/scil/sci_base_domain.c					optional isci
 dev/isci/scil/sci_base_iterator.c				optional isci
 dev/isci/scil/sci_base_library.c				optional isci
 dev/isci/scil/sci_base_logger.c					optional isci
 dev/isci/scil/sci_base_memory_descriptor_list.c			optional isci
 dev/isci/scil/sci_base_memory_descriptor_list_decorator.c	optional isci
 dev/isci/scil/sci_base_object.c					optional isci
 dev/isci/scil/sci_base_observer.c				optional isci
 dev/isci/scil/sci_base_phy.c					optional isci
 dev/isci/scil/sci_base_port.c					optional isci
 dev/isci/scil/sci_base_remote_device.c				optional isci
 dev/isci/scil/sci_base_request.c				optional isci
 dev/isci/scil/sci_base_state_machine.c				optional isci
 dev/isci/scil/sci_base_state_machine_logger.c			optional isci
 dev/isci/scil/sci_base_state_machine_observer.c			optional isci
 dev/isci/scil/sci_base_subject.c				optional isci
 dev/isci/scil/sci_util.c					optional isci
 dev/isci/scil/scic_sds_controller.c				optional isci
 dev/isci/scil/scic_sds_library.c				optional isci
 dev/isci/scil/scic_sds_pci.c					optional isci
 dev/isci/scil/scic_sds_phy.c					optional isci
 dev/isci/scil/scic_sds_port.c					optional isci
 dev/isci/scil/scic_sds_port_configuration_agent.c		optional isci
 dev/isci/scil/scic_sds_remote_device.c				optional isci
 dev/isci/scil/scic_sds_remote_node_context.c			optional isci
 dev/isci/scil/scic_sds_remote_node_table.c			optional isci
 dev/isci/scil/scic_sds_request.c				optional isci
 dev/isci/scil/scic_sds_sgpio.c					optional isci
 dev/isci/scil/scic_sds_smp_remote_device.c			optional isci
 dev/isci/scil/scic_sds_smp_request.c				optional isci
 dev/isci/scil/scic_sds_ssp_request.c				optional isci
 dev/isci/scil/scic_sds_stp_packet_request.c			optional isci
 dev/isci/scil/scic_sds_stp_remote_device.c			optional isci
 dev/isci/scil/scic_sds_stp_request.c				optional isci
 dev/isci/scil/scic_sds_unsolicited_frame_control.c		optional isci
 dev/isci/scil/scif_sas_controller.c				optional isci
 dev/isci/scil/scif_sas_controller_state_handlers.c		optional isci
 dev/isci/scil/scif_sas_controller_states.c			optional isci
 dev/isci/scil/scif_sas_domain.c					optional isci
 dev/isci/scil/scif_sas_domain_state_handlers.c			optional isci
 dev/isci/scil/scif_sas_domain_states.c				optional isci
 dev/isci/scil/scif_sas_high_priority_request_queue.c		optional isci
 dev/isci/scil/scif_sas_internal_io_request.c			optional isci
 dev/isci/scil/scif_sas_io_request.c				optional isci
 dev/isci/scil/scif_sas_io_request_state_handlers.c		optional isci
 dev/isci/scil/scif_sas_io_request_states.c			optional isci
 dev/isci/scil/scif_sas_library.c				optional isci
 dev/isci/scil/scif_sas_remote_device.c				optional isci
 dev/isci/scil/scif_sas_remote_device_ready_substate_handlers.c	optional isci
 dev/isci/scil/scif_sas_remote_device_ready_substates.c		optional isci
 dev/isci/scil/scif_sas_remote_device_starting_substate_handlers.c		optional isci
 dev/isci/scil/scif_sas_remote_device_starting_substates.c	optional isci
 dev/isci/scil/scif_sas_remote_device_state_handlers.c		optional isci
 dev/isci/scil/scif_sas_remote_device_states.c			optional isci
 dev/isci/scil/scif_sas_request.c				optional isci
 dev/isci/scil/scif_sas_smp_activity_clear_affiliation.c		optional isci
 dev/isci/scil/scif_sas_smp_io_request.c				optional isci
 dev/isci/scil/scif_sas_smp_phy.c				optional isci
 dev/isci/scil/scif_sas_smp_remote_device.c			optional isci
 dev/isci/scil/scif_sas_stp_io_request.c				optional isci
 dev/isci/scil/scif_sas_stp_remote_device.c			optional isci
 dev/isci/scil/scif_sas_stp_task_request.c			optional isci
 dev/isci/scil/scif_sas_task_request.c				optional isci
 dev/isci/scil/scif_sas_task_request_state_handlers.c		optional isci
 dev/isci/scil/scif_sas_task_request_states.c			optional isci
 dev/isci/scil/scif_sas_timer.c					optional isci
 isa/syscons_isa.c		optional	sc
 isa/vga_isa.c			optional	vga
 kern/kern_clocksource.c		standard
 kern/link_elf_obj.c		standard
 #
 # IA32 binary support
 #
 #amd64/ia32/ia32_exception.S	optional	compat_freebsd32
 amd64/ia32/ia32_reg.c		optional	compat_freebsd32
 amd64/ia32/ia32_signal.c	optional	compat_freebsd32
 amd64/ia32/ia32_sigtramp.S	optional	compat_freebsd32
 amd64/ia32/ia32_syscall.c	optional	compat_freebsd32
 amd64/ia32/ia32_misc.c		optional	compat_freebsd32
 compat/ia32/ia32_sysvec.c	optional	compat_freebsd32
 compat/linprocfs/linprocfs.c	optional	linprocfs
 compat/linsysfs/linsysfs.c	optional	linsysfs
 #
 # Linux/i386 binary support
 #
 amd64/linux32/linux32_dummy.c	optional	compat_linux32
 amd64/linux32/linux32_locore.s	optional	compat_linux32		\
 	dependency 	"linux32_assym.h"
 amd64/linux32/linux32_machdep.c	optional	compat_linux32
 amd64/linux32/linux32_support.s	optional	compat_linux32		\
 	dependency 	"linux32_assym.h"
 amd64/linux32/linux32_sysent.c	optional	compat_linux32
 amd64/linux32/linux32_sysvec.c	optional	compat_linux32
 compat/linux/linux_emul.c	optional	compat_linux32
 compat/linux/linux_file.c	optional	compat_linux32
 compat/linux/linux_fork.c	optional	compat_linux32
 compat/linux/linux_futex.c	optional	compat_linux32
 compat/linux/linux_getcwd.c	optional	compat_linux32
 compat/linux/linux_ioctl.c	optional	compat_linux32
 compat/linux/linux_ipc.c	optional	compat_linux32
 compat/linux/linux_mib.c	optional	compat_linux32
 compat/linux/linux_misc.c	optional	compat_linux32
 compat/linux/linux_signal.c	optional	compat_linux32
 compat/linux/linux_socket.c	optional	compat_linux32
 compat/linux/linux_stats.c	optional	compat_linux32
 compat/linux/linux_sysctl.c	optional	compat_linux32
 compat/linux/linux_time.c	optional	compat_linux32
 compat/linux/linux_timer.c	optional	compat_linux32
 compat/linux/linux_uid16.c	optional	compat_linux32
 compat/linux/linux_util.c	optional	compat_linux32
 dev/amr/amr_linux.c		optional	compat_linux32 amr
 dev/mfi/mfi_linux.c		optional	compat_linux32 mfi
 #
 # Windows NDIS driver support
 #
 compat/ndis/kern_ndis.c		optional	ndisapi pci
 compat/ndis/kern_windrv.c	optional	ndisapi pci
 compat/ndis/subr_hal.c		optional	ndisapi pci
 compat/ndis/subr_ndis.c		optional	ndisapi pci
 compat/ndis/subr_ntoskrnl.c	optional	ndisapi pci
 compat/ndis/subr_pe.c		optional	ndisapi pci
 compat/ndis/subr_usbd.c		optional	ndisapi pci
 compat/ndis/winx64_wrap.S	optional	ndisapi pci
 #
 libkern/memmove.c		standard
 libkern/memset.c		standard
 #
 # x86 real mode BIOS emulator, required by atkbdc/dpms/vesa
 #
 compat/x86bios/x86bios.c	optional x86bios | atkbd | dpms | vesa
 contrib/x86emu/x86emu.c		optional x86bios | atkbd | dpms | vesa
 #
 # bvm console
 #
 dev/bvm/bvm_console.c		optional	bvmconsole
 dev/bvm/bvm_dbg.c		optional	bvmdebug
 #
 # x86 shared code between IA32, AMD64 and PC98 architectures
 #
 x86/acpica/OsdEnvironment.c	optional	acpi
 x86/acpica/acpi_apm.c		optional	acpi
 x86/acpica/acpi_wakeup.c	optional	acpi
 x86/acpica/madt.c		optional	acpi
 x86/acpica/srat.c		optional	acpi
 x86/bios/smbios.c		optional	smbios
 x86/bios/vpd.c			optional	vpd
 x86/cpufreq/powernow.c		optional	cpufreq
 x86/cpufreq/est.c		optional	cpufreq
 x86/cpufreq/hwpstate.c		optional	cpufreq
 x86/cpufreq/p4tcc.c		optional	cpufreq
 x86/iommu/busdma_dmar.c		optional	acpi acpi_dmar pci
 x86/iommu/intel_ctx.c		optional	acpi acpi_dmar pci
 x86/iommu/intel_drv.c		optional	acpi acpi_dmar pci
 x86/iommu/intel_fault.c		optional	acpi acpi_dmar pci
 x86/iommu/intel_gas.c		optional	acpi acpi_dmar pci
 x86/iommu/intel_idpgtbl.c	optional	acpi acpi_dmar pci
 x86/iommu/intel_qi.c		optional	acpi acpi_dmar pci
 x86/iommu/intel_quirks.c	optional	acpi acpi_dmar pci
 x86/iommu/intel_utils.c		optional	acpi acpi_dmar pci
 x86/isa/atpic.c			optional	atpic isa
 x86/isa/atrtc.c			standard
 x86/isa/clock.c			standard
 x86/isa/elcr.c			optional	atpic isa | mptable
 x86/isa/isa.c			standard
 x86/isa/isa_dma.c		standard
 x86/isa/nmi.c			standard
 x86/isa/orm.c			optional	isa
 x86/pci/pci_bus.c		optional	pci
 x86/pci/qpi.c			optional	pci
 x86/x86/busdma_bounce.c		standard
 x86/x86/busdma_machdep.c	standard
 x86/x86/dump_machdep.c		standard
 x86/x86/fdt_machdep.c		optional	fdt
 x86/x86/identcpu.c		standard
 x86/x86/intr_machdep.c		standard
 x86/x86/io_apic.c		standard
 x86/x86/legacy.c		standard
 x86/x86/local_apic.c		standard
 x86/x86/mca.c			standard
 x86/x86/mptable.c		optional	mptable
 x86/x86/mptable_pci.c		optional	mptable pci
 x86/x86/msi.c			optional	pci
 x86/x86/nexus.c			standard
 x86/x86/tsc.c			standard
 x86/x86/delay.c			standard
 x86/xen/hvm.c			optional	xenhvm
 x86/xen/xen_intr.c		optional	xen | xenhvm
 x86/xen/pv.c			optional	xenhvm
 x86/xen/pvcpu_enum.c		optional	xenhvm
 x86/xen/xen_apic.c		optional	xenhvm
 x86/xen/xenpv.c			optional	xenhvm
 x86/xen/xen_nexus.c		optional	xenhvm
 x86/xen/xen_msi.c		optional	xenhvm
 x86/xen/xen_pci_bus.c		optional	xenhvm
Index: projects/clang360-import/sys/conf/kern.mk
===================================================================
--- projects/clang360-import/sys/conf/kern.mk	(revision 277944)
+++ projects/clang360-import/sys/conf/kern.mk	(revision 277945)
@@ -1,209 +1,210 @@
 # $FreeBSD$
 
 #
 # Warning flags for compiling the kernel and components of the kernel:
 #
 CWARNFLAGS?=	-Wall -Wredundant-decls -Wnested-externs -Wstrict-prototypes \
 		-Wmissing-prototypes -Wpointer-arith -Winline -Wcast-qual \
 		-Wundef -Wno-pointer-sign ${FORMAT_EXTENSIONS} \
 		-Wmissing-include-dirs -fdiagnostics-show-option \
 		-Wno-unknown-pragmas \
 		${CWARNEXTRA}
 #
 # The following flags are next up for working on:
 #	-Wextra
 
 # Disable a few warnings for clang, since there are several places in the
 # kernel where fixing them is more trouble than it is worth, or where there is
 # a false positive.
 .if ${COMPILER_TYPE} == "clang"
 NO_WCONSTANT_CONVERSION=	-Wno-constant-conversion
 NO_WSHIFT_COUNT_NEGATIVE=	-Wno-shift-count-negative
 NO_WSHIFT_COUNT_OVERFLOW=	-Wno-shift-count-overflow
 NO_WSELF_ASSIGN=		-Wno-self-assign
 NO_WUNNEEDED_INTERNAL_DECL=	-Wno-unneeded-internal-declaration
 NO_WSOMETIMES_UNINITIALIZED=	-Wno-error-sometimes-uninitialized
+NO_WCAST_QUAL=			-Wno-cast-qual
 # Several other warnings which might be useful in some cases, but not severe
 # enough to error out the whole kernel build.  Display them anyway, so there is
 # some incentive to fix them eventually.
 CWARNEXTRA?=	-Wno-error-tautological-compare -Wno-error-empty-body \
 		-Wno-error-parentheses-equality -Wno-error-unused-function \
 		-Wno-error-pointer-sign
 
 CLANG_NO_IAS= -no-integrated-as
 .if ${COMPILER_VERSION} < 30500
 # XXX: clang < 3.5 integrated-as doesn't grok .codeNN directives
 CLANG_NO_IAS34= -no-integrated-as
 .endif
 .endif
 
 .if ${COMPILER_TYPE} == "gcc"
 GCC_MS_EXTENSIONS= -fms-extensions
 .if ${COMPILER_VERSION} >= 40300
 # Catch-all for all the things that are in our tree, but for which we're
 # not yet ready for this compiler. Note: we likely only really "support"
 # building with gcc 4.8 and newer. Nothing older has been tested.
 CWARNEXTRA?=	-Wno-error=inline -Wno-error=enum-compare -Wno-error=unused-but-set-variable \
 		-Wno-error=aggressive-loop-optimizations -Wno-error=maybe-uninitialized \
 		-Wno-error=array-bounds -Wno-error=address \
 		-Wno-error=cast-qual -Wno-error=sequence-point -Wno-error=attributes \
 		-Wno-error=strict-overflow -Wno-error=overflow
 .else
 # For gcc 4.2, eliminate the too-often-wrong warnings about uninitialized vars.
 CWARNEXTRA?=	-Wno-uninitialized
 .endif
 .endif
 
 # External compilers may not support our format extensions.  Allow them
 # to be disabled.  WARNING: format checking is disabled in this case.
 .if ${MK_FORMAT_EXTENSIONS} == "no"
 FORMAT_EXTENSIONS=	-Wno-format
 .elif ${COMPILER_TYPE} == "clang" && ${COMPILER_VERSION} >= 30600
 FORMAT_EXTENSIONS=	-D__printf__=__freebsd_kprintf__
 .else
 FORMAT_EXTENSIONS=	-fformat-extensions
 .endif
 
 #
 # On i386, do not align the stack to 16-byte boundaries.  Otherwise GCC 2.95
 # and above adds code to the entry and exit point of every function to align the
 # stack to 16-byte boundaries -- thus wasting approximately 12 bytes of stack
 # per function call.  While the 16-byte alignment may benefit micro benchmarks,
 # it is probably an overall loss as it makes the code bigger (less efficient
 # use of code cache tag lines) and uses more stack (less efficient use of data
 # cache tag lines).  Explicitly prohibit the use of FPU, SSE and other SIMD
 # operations inside the kernel itself.  These operations are exclusively
 # reserved for user applications.
 #
 # gcc:
 # Setting -mno-mmx implies -mno-3dnow
 # Setting -mno-sse implies -mno-sse2, -mno-sse3 and -mno-ssse3
 #
 # clang:
 # Setting -mno-mmx implies -mno-3dnow and -mno-3dnowa
 # Setting -mno-sse implies -mno-sse2, -mno-sse3, -mno-ssse3, -mno-sse41 and -mno-sse42
 #
 .if ${MACHINE_CPUARCH} == "i386"
 CFLAGS.gcc+=	-mno-align-long-strings -mpreferred-stack-boundary=2
 CFLAGS.clang+=	-mno-aes -mno-avx
 CFLAGS+=	-mno-mmx -mno-sse -msoft-float
 INLINE_LIMIT?=	8000
 .endif
 
 .if ${MACHINE_CPUARCH} == "arm"
 INLINE_LIMIT?=	8000
 .endif
 
 #
 # For sparc64 we want the medany code model so modules may be located
 # anywhere in the 64-bit address space.  We also tell GCC to use floating
 # point emulation.  This avoids using floating point registers for integer
 # operations which it has a tendency to do.
 #
 .if ${MACHINE_CPUARCH} == "sparc64"
 CFLAGS.clang+=	-mcmodel=large -fno-dwarf2-cfi-asm
 CFLAGS.gcc+=	-mcmodel=medany -msoft-float
 INLINE_LIMIT?=	15000
 .endif
 
 #
 # For AMD64, we explicitly prohibit the use of FPU, SSE and other SIMD
 # operations inside the kernel itself.  These operations are exclusively
 # reserved for user applications.
 #
 # gcc:
 # Setting -mno-mmx implies -mno-3dnow
 # Setting -mno-sse implies -mno-sse2, -mno-sse3, -mno-ssse3 and -mfpmath=387
 #
 # clang:
 # Setting -mno-mmx implies -mno-3dnow and -mno-3dnowa
 # Setting -mno-sse implies -mno-sse2, -mno-sse3, -mno-ssse3, -mno-sse41 and -mno-sse42
 # (-mfpmath= is not supported)
 #
 .if ${MACHINE_CPUARCH} == "amd64"
 CFLAGS.clang+=	-mno-aes -mno-avx
 CFLAGS+=	-mcmodel=kernel -mno-red-zone -mno-mmx -mno-sse -msoft-float \
 		-fno-asynchronous-unwind-tables
 INLINE_LIMIT?=	8000
 .endif
 
 #
 # For PowerPC we tell gcc to use floating point emulation.  This avoids using
 # floating point registers for integer operations which it has a tendency to do.
 # Also explicitly disable Altivec instructions inside the kernel.
 #
 .if ${MACHINE_CPUARCH} == "powerpc"
 CFLAGS+=	-msoft-float -mno-altivec
 INLINE_LIMIT?=	15000
 .endif
 
 #
 # Use dot symbols on powerpc64 to make ddb happy
 #
 .if ${MACHINE_ARCH} == "powerpc64"
 CFLAGS+=	-mcall-aixdesc
 .endif
 
 #
 # For MIPS we also tell gcc to use floating point emulation
 #
 .if ${MACHINE_CPUARCH} == "mips"
 CFLAGS+=	-msoft-float
 INLINE_LIMIT?=	8000
 .endif
 
 #
 # GCC 3.0 and above like to do certain optimizations based on the
 # assumption that the program is linked against libc.  Stop this.
 #
 CFLAGS+=	-ffreestanding
 
 #
 # GCC SSP support
 #
 .if ${MK_SSP} != "no" && \
     ${MACHINE_CPUARCH} != "arm" && ${MACHINE_CPUARCH} != "mips"
 CFLAGS+=	-fstack-protector
 .endif
 
 #
 # Add -gdwarf-2 when compiling -g. The default starting in clang v3.4
 # and gcc 4.8 is to generate DWARF version 4. However, our tools don't
 # cope well with DWARF 4, so force it to genereate DWARF2, which they
 # understand. Do this unconditionally as it is harmless when not needed,
 # but critical for these newer versions.
 #
 .if ${CFLAGS:M-g} != "" && ${CFLAGS:M-gdwarf*} == ""
 CFLAGS+=	-gdwarf-2
 .endif
 
 CFLAGS+= ${CWARNEXTRA} ${CWARNFLAGS} ${CWARNFLAGS.${.IMPSRC:T}}
 CFLAGS+= ${CFLAGS.${COMPILER_TYPE}} ${CFLAGS.${.IMPSRC:T}}
 
 # Tell bmake not to mistake standard targets for things to be searched for
 # or expect to ever be up-to-date.
 PHONY_NOTMAIN = afterdepend afterinstall all beforedepend beforeinstall \
 		beforelinking build build-tools buildfiles buildincludes \
 		checkdpadd clean cleandepend cleandir cleanobj configure \
 		depend dependall distclean distribute exe \
 		html includes install installfiles installincludes lint \
 		obj objlink objs objwarn realall realdepend \
 		realinstall regress subdir-all subdir-depend subdir-install \
 		tags whereobj
 
 .PHONY: ${PHONY_NOTMAIN}
 .NOTMAIN: ${PHONY_NOTMAIN}
 
 CSTD=		c99
 
 .if ${CSTD} == "k&r"
 CFLAGS+=        -traditional
 .elif ${CSTD} == "c89" || ${CSTD} == "c90"
 CFLAGS+=        -std=iso9899:1990
 .elif ${CSTD} == "c94" || ${CSTD} == "c95"
 CFLAGS+=        -std=iso9899:199409
 .elif ${CSTD} == "c99"
 CFLAGS+=        -std=iso9899:1999
 .else # CSTD
 CFLAGS+=        -std=${CSTD}
 .endif # CSTD
Index: projects/clang360-import/sys/conf
===================================================================
--- projects/clang360-import/sys/conf	(revision 277944)
+++ projects/clang360-import/sys/conf	(revision 277945)

Property changes on: projects/clang360-import/sys/conf
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/sys/conf:r277844-277944
Index: projects/clang360-import/sys/dev/alc/if_alc.c
===================================================================
--- projects/clang360-import/sys/dev/alc/if_alc.c	(revision 277944)
+++ projects/clang360-import/sys/dev/alc/if_alc.c	(revision 277945)
@@ -1,4621 +1,4621 @@
 /*-
  * Copyright (c) 2009, Pyun YongHyeon <yongari@FreeBSD.org>
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice unmodified, this list of conditions, and the following
  *    disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 /* Driver for Atheros AR813x/AR815x PCIe Ethernet. */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/bus.h>
 #include <sys/endian.h>
 #include <sys/kernel.h>
 #include <sys/lock.h>
 #include <sys/malloc.h>
 #include <sys/mbuf.h>
 #include <sys/module.h>
 #include <sys/mutex.h>
 #include <sys/rman.h>
 #include <sys/queue.h>
 #include <sys/socket.h>
 #include <sys/sockio.h>
 #include <sys/sysctl.h>
 #include <sys/taskqueue.h>
 
 #include <net/bpf.h>
 #include <net/if.h>
 #include <net/if_var.h>
 #include <net/if_arp.h>
 #include <net/ethernet.h>
 #include <net/if_dl.h>
 #include <net/if_llc.h>
 #include <net/if_media.h>
 #include <net/if_types.h>
 #include <net/if_vlan_var.h>
 
 #include <netinet/in.h>
 #include <netinet/in_systm.h>
 #include <netinet/ip.h>
 #include <netinet/tcp.h>
 
 #include <dev/mii/mii.h>
 #include <dev/mii/miivar.h>
 
 #include <dev/pci/pcireg.h>
 #include <dev/pci/pcivar.h>
 
 #include <machine/bus.h>
 #include <machine/in_cksum.h>
 
 #include <dev/alc/if_alcreg.h>
 #include <dev/alc/if_alcvar.h>
 
 /* "device miibus" required.  See GENERIC if you get errors here. */
 #include "miibus_if.h"
 #undef ALC_USE_CUSTOM_CSUM
 
 #ifdef ALC_USE_CUSTOM_CSUM
 #define	ALC_CSUM_FEATURES	(CSUM_TCP | CSUM_UDP)
 #else
 #define	ALC_CSUM_FEATURES	(CSUM_IP | CSUM_TCP | CSUM_UDP)
 #endif
 
 MODULE_DEPEND(alc, pci, 1, 1, 1);
 MODULE_DEPEND(alc, ether, 1, 1, 1);
 MODULE_DEPEND(alc, miibus, 1, 1, 1);
 
 /* Tunables. */
 static int msi_disable = 0;
 static int msix_disable = 0;
 TUNABLE_INT("hw.alc.msi_disable", &msi_disable);
 TUNABLE_INT("hw.alc.msix_disable", &msix_disable);
 
 /*
  * Devices supported by this driver.
  */
 static struct alc_ident alc_ident_table[] = {
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8131, 9 * 1024,
 		"Atheros AR8131 PCIe Gigabit Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8132, 9 * 1024,
 		"Atheros AR8132 PCIe Fast Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8151, 6 * 1024,
 		"Atheros AR8151 v1.0 PCIe Gigabit Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8151_V2, 6 * 1024,
 		"Atheros AR8151 v2.0 PCIe Gigabit Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8152_B, 6 * 1024,
 		"Atheros AR8152 v1.1 PCIe Fast Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8152_B2, 6 * 1024,
 		"Atheros AR8152 v2.0 PCIe Fast Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8161, 9 * 1024,
 		"Atheros AR8161 PCIe Gigabit Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8162, 9 * 1024,
-		"Atheros AR8161 PCIe Fast Ethernet" },
+		"Atheros AR8162 PCIe Fast Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8171, 9 * 1024,
-		"Atheros AR8161 PCIe Gigabit Ethernet" },
+		"Atheros AR8171 PCIe Gigabit Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_AR8172, 9 * 1024,
-		"Atheros AR8161 PCIe Fast Ethernet" },
+		"Atheros AR8172 PCIe Fast Ethernet" },
 	{ VENDORID_ATHEROS, DEVICEID_ATHEROS_E2200, 9 * 1024,
 		"Killer E2200 Gigabit Ethernet" },
 	{ 0, 0, 0, NULL}
 };
 
 static void	alc_aspm(struct alc_softc *, int, int);
 static void	alc_aspm_813x(struct alc_softc *, int);
 static void	alc_aspm_816x(struct alc_softc *, int);
 static int	alc_attach(device_t);
 static int	alc_check_boundary(struct alc_softc *);
 static void	alc_config_msi(struct alc_softc *);
 static int	alc_detach(device_t);
 static void	alc_disable_l0s_l1(struct alc_softc *);
 static int	alc_dma_alloc(struct alc_softc *);
 static void	alc_dma_free(struct alc_softc *);
 static void	alc_dmamap_cb(void *, bus_dma_segment_t *, int, int);
 static void	alc_dsp_fixup(struct alc_softc *, int);
 static int	alc_encap(struct alc_softc *, struct mbuf **);
 static struct alc_ident *
 		alc_find_ident(device_t);
 #ifndef __NO_STRICT_ALIGNMENT
 static struct mbuf *
 		alc_fixup_rx(struct ifnet *, struct mbuf *);
 #endif
 static void	alc_get_macaddr(struct alc_softc *);
 static void	alc_get_macaddr_813x(struct alc_softc *);
 static void	alc_get_macaddr_816x(struct alc_softc *);
 static void	alc_get_macaddr_par(struct alc_softc *);
 static void	alc_init(void *);
 static void	alc_init_cmb(struct alc_softc *);
 static void	alc_init_locked(struct alc_softc *);
 static void	alc_init_rr_ring(struct alc_softc *);
 static int	alc_init_rx_ring(struct alc_softc *);
 static void	alc_init_smb(struct alc_softc *);
 static void	alc_init_tx_ring(struct alc_softc *);
 static void	alc_int_task(void *, int);
 static int	alc_intr(void *);
 static int	alc_ioctl(struct ifnet *, u_long, caddr_t);
 static void	alc_mac_config(struct alc_softc *);
 static uint32_t	alc_mii_readreg_813x(struct alc_softc *, int, int);
 static uint32_t	alc_mii_readreg_816x(struct alc_softc *, int, int);
 static uint32_t	alc_mii_writereg_813x(struct alc_softc *, int, int, int);
 static uint32_t	alc_mii_writereg_816x(struct alc_softc *, int, int, int);
 static int	alc_miibus_readreg(device_t, int, int);
 static void	alc_miibus_statchg(device_t);
 static int	alc_miibus_writereg(device_t, int, int, int);
 static uint32_t	alc_miidbg_readreg(struct alc_softc *, int);
 static uint32_t	alc_miidbg_writereg(struct alc_softc *, int, int);
 static uint32_t	alc_miiext_readreg(struct alc_softc *, int, int);
 static uint32_t	alc_miiext_writereg(struct alc_softc *, int, int, int);
 static int	alc_mediachange(struct ifnet *);
 static int	alc_mediachange_locked(struct alc_softc *);
 static void	alc_mediastatus(struct ifnet *, struct ifmediareq *);
 static int	alc_newbuf(struct alc_softc *, struct alc_rxdesc *);
 static void	alc_osc_reset(struct alc_softc *);
 static void	alc_phy_down(struct alc_softc *);
 static void	alc_phy_reset(struct alc_softc *);
 static void	alc_phy_reset_813x(struct alc_softc *);
 static void	alc_phy_reset_816x(struct alc_softc *);
 static int	alc_probe(device_t);
 static void	alc_reset(struct alc_softc *);
 static int	alc_resume(device_t);
 static void	alc_rxeof(struct alc_softc *, struct rx_rdesc *);
 static int	alc_rxintr(struct alc_softc *, int);
 static void	alc_rxfilter(struct alc_softc *);
 static void	alc_rxvlan(struct alc_softc *);
 static void	alc_setlinkspeed(struct alc_softc *);
 static void	alc_setwol(struct alc_softc *);
 static void	alc_setwol_813x(struct alc_softc *);
 static void	alc_setwol_816x(struct alc_softc *);
 static int	alc_shutdown(device_t);
 static void	alc_start(struct ifnet *);
 static void	alc_start_locked(struct ifnet *);
 static void	alc_start_queue(struct alc_softc *);
 static void	alc_stats_clear(struct alc_softc *);
 static void	alc_stats_update(struct alc_softc *);
 static void	alc_stop(struct alc_softc *);
 static void	alc_stop_mac(struct alc_softc *);
 static void	alc_stop_queue(struct alc_softc *);
 static int	alc_suspend(device_t);
 static void	alc_sysctl_node(struct alc_softc *);
 static void	alc_tick(void *);
 static void	alc_txeof(struct alc_softc *);
 static void	alc_watchdog(struct alc_softc *);
 static int	sysctl_int_range(SYSCTL_HANDLER_ARGS, int, int);
 static int	sysctl_hw_alc_proc_limit(SYSCTL_HANDLER_ARGS);
 static int	sysctl_hw_alc_int_mod(SYSCTL_HANDLER_ARGS);
 
 static device_method_t alc_methods[] = {
 	/* Device interface. */
 	DEVMETHOD(device_probe,		alc_probe),
 	DEVMETHOD(device_attach,	alc_attach),
 	DEVMETHOD(device_detach,	alc_detach),
 	DEVMETHOD(device_shutdown,	alc_shutdown),
 	DEVMETHOD(device_suspend,	alc_suspend),
 	DEVMETHOD(device_resume,	alc_resume),
 
 	/* MII interface. */
 	DEVMETHOD(miibus_readreg,	alc_miibus_readreg),
 	DEVMETHOD(miibus_writereg,	alc_miibus_writereg),
 	DEVMETHOD(miibus_statchg,	alc_miibus_statchg),
 
 	{ NULL, NULL }
 };
 
 static driver_t alc_driver = {
 	"alc",
 	alc_methods,
 	sizeof(struct alc_softc)
 };
 
 static devclass_t alc_devclass;
 
 DRIVER_MODULE(alc, pci, alc_driver, alc_devclass, 0, 0);
 DRIVER_MODULE(miibus, alc, miibus_driver, miibus_devclass, 0, 0);
 
 static struct resource_spec alc_res_spec_mem[] = {
 	{ SYS_RES_MEMORY,	PCIR_BAR(0),	RF_ACTIVE },
 	{ -1,			0,		0 }
 };
 
 static struct resource_spec alc_irq_spec_legacy[] = {
 	{ SYS_RES_IRQ,		0,		RF_ACTIVE | RF_SHAREABLE },
 	{ -1,			0,		0 }
 };
 
 static struct resource_spec alc_irq_spec_msi[] = {
 	{ SYS_RES_IRQ,		1,		RF_ACTIVE },
 	{ -1,			0,		0 }
 };
 
 static struct resource_spec alc_irq_spec_msix[] = {
 	{ SYS_RES_IRQ,		1,		RF_ACTIVE },
 	{ -1,			0,		0 }
 };
 
 static uint32_t alc_dma_burst[] = { 128, 256, 512, 1024, 2048, 4096, 0 };
 
 static int
 alc_miibus_readreg(device_t dev, int phy, int reg)
 {
 	struct alc_softc *sc;
 	int v;
 
 	sc = device_get_softc(dev);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		v = alc_mii_readreg_816x(sc, phy, reg);
 	else
 		v = alc_mii_readreg_813x(sc, phy, reg);
 	return (v);
 }
 
 static uint32_t
 alc_mii_readreg_813x(struct alc_softc *sc, int phy, int reg)
 {
 	uint32_t v;
 	int i;
 
 	/*
 	 * For AR8132 fast ethernet controller, do not report 1000baseT
 	 * capability to mii(4). Even though AR8132 uses the same
 	 * model/revision number of F1 gigabit PHY, the PHY has no
 	 * ability to establish 1000baseT link.
 	 */
 	if ((sc->alc_flags & ALC_FLAG_FASTETHER) != 0 &&
 	    reg == MII_EXTSR)
 		return (0);
 
 	CSR_WRITE_4(sc, ALC_MDIO, MDIO_OP_EXECUTE | MDIO_OP_READ |
 	    MDIO_SUP_PREAMBLE | MDIO_CLK_25_4 | MDIO_REG_ADDR(reg));
 	for (i = ALC_PHY_TIMEOUT; i > 0; i--) {
 		DELAY(5);
 		v = CSR_READ_4(sc, ALC_MDIO);
 		if ((v & (MDIO_OP_EXECUTE | MDIO_OP_BUSY)) == 0)
 			break;
 	}
 
 	if (i == 0) {
 		device_printf(sc->alc_dev, "phy read timeout : %d\n", reg);
 		return (0);
 	}
 
 	return ((v & MDIO_DATA_MASK) >> MDIO_DATA_SHIFT);
 }
 
 static uint32_t
 alc_mii_readreg_816x(struct alc_softc *sc, int phy, int reg)
 {
 	uint32_t clk, v;
 	int i;
 
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0)
 		clk = MDIO_CLK_25_128;
 	else
 		clk = MDIO_CLK_25_4;
 	CSR_WRITE_4(sc, ALC_MDIO, MDIO_OP_EXECUTE | MDIO_OP_READ |
 	    MDIO_SUP_PREAMBLE | clk | MDIO_REG_ADDR(reg));
 	for (i = ALC_PHY_TIMEOUT; i > 0; i--) {
 		DELAY(5);
 		v = CSR_READ_4(sc, ALC_MDIO);
 		if ((v & MDIO_OP_BUSY) == 0)
 			break;
 	}
 
 	if (i == 0) {
 		device_printf(sc->alc_dev, "phy read timeout : %d\n", reg);
 		return (0);
 	}
 
 	return ((v & MDIO_DATA_MASK) >> MDIO_DATA_SHIFT);
 }
 
 static int
 alc_miibus_writereg(device_t dev, int phy, int reg, int val)
 {
 	struct alc_softc *sc;
 	int v;
 
 	sc = device_get_softc(dev);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		v = alc_mii_writereg_816x(sc, phy, reg, val);
 	else
 		v = alc_mii_writereg_813x(sc, phy, reg, val);
 	return (v);
 }
 
 static uint32_t
 alc_mii_writereg_813x(struct alc_softc *sc, int phy, int reg, int val)
 {
 	uint32_t v;
 	int i;
 
 	CSR_WRITE_4(sc, ALC_MDIO, MDIO_OP_EXECUTE | MDIO_OP_WRITE |
 	    (val & MDIO_DATA_MASK) << MDIO_DATA_SHIFT |
 	    MDIO_SUP_PREAMBLE | MDIO_CLK_25_4 | MDIO_REG_ADDR(reg));
 	for (i = ALC_PHY_TIMEOUT; i > 0; i--) {
 		DELAY(5);
 		v = CSR_READ_4(sc, ALC_MDIO);
 		if ((v & (MDIO_OP_EXECUTE | MDIO_OP_BUSY)) == 0)
 			break;
 	}
 
 	if (i == 0)
 		device_printf(sc->alc_dev, "phy write timeout : %d\n", reg);
 
 	return (0);
 }
 
 static uint32_t
 alc_mii_writereg_816x(struct alc_softc *sc, int phy, int reg, int val)
 {
 	uint32_t clk, v;
 	int i;
 
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0)
 		clk = MDIO_CLK_25_128;
 	else
 		clk = MDIO_CLK_25_4;
 	CSR_WRITE_4(sc, ALC_MDIO, MDIO_OP_EXECUTE | MDIO_OP_WRITE |
 	    ((val & MDIO_DATA_MASK) << MDIO_DATA_SHIFT) | MDIO_REG_ADDR(reg) |
 	    MDIO_SUP_PREAMBLE | clk);
 	for (i = ALC_PHY_TIMEOUT; i > 0; i--) {
 		DELAY(5);
 		v = CSR_READ_4(sc, ALC_MDIO);
 		if ((v & MDIO_OP_BUSY) == 0)
 			break;
 	}
 
 	if (i == 0)
 		device_printf(sc->alc_dev, "phy write timeout : %d\n", reg);
 
 	return (0);
 }
 
 static void
 alc_miibus_statchg(device_t dev)
 {
 	struct alc_softc *sc;
 	struct mii_data *mii;
 	struct ifnet *ifp;
 	uint32_t reg;
 
 	sc = device_get_softc(dev);
 
 	mii = device_get_softc(sc->alc_miibus);
 	ifp = sc->alc_ifp;
 	if (mii == NULL || ifp == NULL ||
 	    (ifp->if_drv_flags & IFF_DRV_RUNNING) == 0)
 		return;
 
 	sc->alc_flags &= ~ALC_FLAG_LINK;
 	if ((mii->mii_media_status & (IFM_ACTIVE | IFM_AVALID)) ==
 	    (IFM_ACTIVE | IFM_AVALID)) {
 		switch (IFM_SUBTYPE(mii->mii_media_active)) {
 		case IFM_10_T:
 		case IFM_100_TX:
 			sc->alc_flags |= ALC_FLAG_LINK;
 			break;
 		case IFM_1000_T:
 			if ((sc->alc_flags & ALC_FLAG_FASTETHER) == 0)
 				sc->alc_flags |= ALC_FLAG_LINK;
 			break;
 		default:
 			break;
 		}
 	}
 	/* Stop Rx/Tx MACs. */
 	alc_stop_mac(sc);
 
 	/* Program MACs with resolved speed/duplex/flow-control. */
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0) {
 		alc_start_queue(sc);
 		alc_mac_config(sc);
 		/* Re-enable Tx/Rx MACs. */
 		reg = CSR_READ_4(sc, ALC_MAC_CFG);
 		reg |= MAC_CFG_TX_ENB | MAC_CFG_RX_ENB;
 		CSR_WRITE_4(sc, ALC_MAC_CFG, reg);
 	}
 	alc_aspm(sc, 0, IFM_SUBTYPE(mii->mii_media_active));
 	alc_dsp_fixup(sc, IFM_SUBTYPE(mii->mii_media_active));
 }
 
 static uint32_t
 alc_miidbg_readreg(struct alc_softc *sc, int reg)
 {
 
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, ALC_MII_DBG_ADDR,
 	    reg);
 	return (alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA));
 }
 
 static uint32_t
 alc_miidbg_writereg(struct alc_softc *sc, int reg, int val)
 {
 
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, ALC_MII_DBG_ADDR,
 	    reg);
 	return (alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA, val));
 }
 
 static uint32_t
 alc_miiext_readreg(struct alc_softc *sc, int devaddr, int reg)
 {
 	uint32_t clk, v;
 	int i;
 
 	CSR_WRITE_4(sc, ALC_EXT_MDIO, EXT_MDIO_REG(reg) |
 	    EXT_MDIO_DEVADDR(devaddr));
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0)
 		clk = MDIO_CLK_25_128;
 	else
 		clk = MDIO_CLK_25_4;
 	CSR_WRITE_4(sc, ALC_MDIO, MDIO_OP_EXECUTE | MDIO_OP_READ |
 	    MDIO_SUP_PREAMBLE | clk | MDIO_MODE_EXT);
 	for (i = ALC_PHY_TIMEOUT; i > 0; i--) {
 		DELAY(5);
 		v = CSR_READ_4(sc, ALC_MDIO);
 		if ((v & MDIO_OP_BUSY) == 0)
 			break;
 	}
 
 	if (i == 0) {
 		device_printf(sc->alc_dev, "phy ext read timeout : %d, %d\n",
 		    devaddr, reg);
 		return (0);
 	}
 
 	return ((v & MDIO_DATA_MASK) >> MDIO_DATA_SHIFT);
 }
 
 static uint32_t
 alc_miiext_writereg(struct alc_softc *sc, int devaddr, int reg, int val)
 {
 	uint32_t clk, v;
 	int i;
 
 	CSR_WRITE_4(sc, ALC_EXT_MDIO, EXT_MDIO_REG(reg) |
 	    EXT_MDIO_DEVADDR(devaddr));
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0)
 		clk = MDIO_CLK_25_128;
 	else
 		clk = MDIO_CLK_25_4;
 	CSR_WRITE_4(sc, ALC_MDIO, MDIO_OP_EXECUTE | MDIO_OP_WRITE |
 	    ((val & MDIO_DATA_MASK) << MDIO_DATA_SHIFT) |
 	    MDIO_SUP_PREAMBLE | clk | MDIO_MODE_EXT);
 	for (i = ALC_PHY_TIMEOUT; i > 0; i--) {
 		DELAY(5);
 		v = CSR_READ_4(sc, ALC_MDIO);
 		if ((v & MDIO_OP_BUSY) == 0)
 			break;
 	}
 
 	if (i == 0)
 		device_printf(sc->alc_dev, "phy ext write timeout : %d, %d\n",
 		    devaddr, reg);
 
 	return (0);
 }
 
 static void
 alc_dsp_fixup(struct alc_softc *sc, int media)
 {
 	uint16_t agc, len, val;
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		return;
 	if (AR816X_REV(sc->alc_rev) >= AR816X_REV_C0)
 		return;
 
 	/*
 	 * Vendor PHY magic.
 	 * 1000BT/AZ, wrong cable length
 	 */
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0) {
 		len = alc_miiext_readreg(sc, MII_EXT_PCS, MII_EXT_CLDCTL6);
 		len = (len >> EXT_CLDCTL6_CAB_LEN_SHIFT) &
 		    EXT_CLDCTL6_CAB_LEN_MASK;
 		agc = alc_miidbg_readreg(sc, MII_DBG_AGC);
 		agc = (agc >> DBG_AGC_2_VGA_SHIFT) & DBG_AGC_2_VGA_MASK;
 		if ((media == IFM_1000_T && len > EXT_CLDCTL6_CAB_LEN_SHORT1G &&
 		    agc > DBG_AGC_LONG1G_LIMT) ||
 		    (media == IFM_100_TX && len > DBG_AGC_LONG100M_LIMT &&
 		    agc > DBG_AGC_LONG1G_LIMT)) {
 			alc_miidbg_writereg(sc, MII_DBG_AZ_ANADECT,
 			    DBG_AZ_ANADECT_LONG);
 			val = alc_miiext_readreg(sc, MII_EXT_ANEG,
 			    MII_EXT_ANEG_AFE);
 			val |= ANEG_AFEE_10BT_100M_TH;
 			alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_AFE,
 			    val);
 		} else {
 			alc_miidbg_writereg(sc, MII_DBG_AZ_ANADECT,
 			    DBG_AZ_ANADECT_DEFAULT);
 			val = alc_miiext_readreg(sc, MII_EXT_ANEG,
 			    MII_EXT_ANEG_AFE);
 			val &= ~ANEG_AFEE_10BT_100M_TH;
 			alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_AFE,
 			    val);
 		}
 		if ((sc->alc_flags & ALC_FLAG_LINK_WAR) != 0 &&
 		    AR816X_REV(sc->alc_rev) == AR816X_REV_B0) {
 			if (media == IFM_1000_T) {
 				/*
 				 * Giga link threshold, raise the tolerance of
 				 * noise 50%.
 				 */
 				val = alc_miidbg_readreg(sc, MII_DBG_MSE20DB);
 				val &= ~DBG_MSE20DB_TH_MASK;
 				val |= (DBG_MSE20DB_TH_HI <<
 				    DBG_MSE20DB_TH_SHIFT);
 				alc_miidbg_writereg(sc, MII_DBG_MSE20DB, val);
 			} else if (media == IFM_100_TX)
 				alc_miidbg_writereg(sc, MII_DBG_MSE16DB,
 				    DBG_MSE16DB_UP);
 		}
 	} else {
 		val = alc_miiext_readreg(sc, MII_EXT_ANEG, MII_EXT_ANEG_AFE);
 		val &= ~ANEG_AFEE_10BT_100M_TH;
 		alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_AFE, val);
 		if ((sc->alc_flags & ALC_FLAG_LINK_WAR) != 0 &&
 		    AR816X_REV(sc->alc_rev) == AR816X_REV_B0) {
 			alc_miidbg_writereg(sc, MII_DBG_MSE16DB,
 			    DBG_MSE16DB_DOWN);
 			val = alc_miidbg_readreg(sc, MII_DBG_MSE20DB);
 			val &= ~DBG_MSE20DB_TH_MASK;
 			val |= (DBG_MSE20DB_TH_DEFAULT << DBG_MSE20DB_TH_SHIFT);
 			alc_miidbg_writereg(sc, MII_DBG_MSE20DB, val);
 		}
 	}
 }
 
 static void
 alc_mediastatus(struct ifnet *ifp, struct ifmediareq *ifmr)
 {
 	struct alc_softc *sc;
 	struct mii_data *mii;
 
 	sc = ifp->if_softc;
 	ALC_LOCK(sc);
 	if ((ifp->if_flags & IFF_UP) == 0) {
 		ALC_UNLOCK(sc);
 		return;
 	}
 	mii = device_get_softc(sc->alc_miibus);
 
 	mii_pollstat(mii);
 	ifmr->ifm_status = mii->mii_media_status;
 	ifmr->ifm_active = mii->mii_media_active;
 	ALC_UNLOCK(sc);
 }
 
 static int
 alc_mediachange(struct ifnet *ifp)
 {
 	struct alc_softc *sc;
 	int error;
 
 	sc = ifp->if_softc;
 	ALC_LOCK(sc);
 	error = alc_mediachange_locked(sc);
 	ALC_UNLOCK(sc);
 
 	return (error);
 }
 
 static int
 alc_mediachange_locked(struct alc_softc *sc)
 {
 	struct mii_data *mii;
 	struct mii_softc *miisc;
 	int error;
 
 	ALC_LOCK_ASSERT(sc);
 
 	mii = device_get_softc(sc->alc_miibus);
 	LIST_FOREACH(miisc, &mii->mii_phys, mii_list)
 		PHY_RESET(miisc);
 	error = mii_mediachg(mii);
 
 	return (error);
 }
 
 static struct alc_ident *
 alc_find_ident(device_t dev)
 {
 	struct alc_ident *ident;
 	uint16_t vendor, devid;
 
 	vendor = pci_get_vendor(dev);
 	devid = pci_get_device(dev);
 	for (ident = alc_ident_table; ident->name != NULL; ident++) {
 		if (vendor == ident->vendorid && devid == ident->deviceid)
 			return (ident);
 	}
 
 	return (NULL);
 }
 
 static int
 alc_probe(device_t dev)
 {
 	struct alc_ident *ident;
 
 	ident = alc_find_ident(dev);
 	if (ident != NULL) {
 		device_set_desc(dev, ident->name);
 		return (BUS_PROBE_DEFAULT);
 	}
 
 	return (ENXIO);
 }
 
 static void
 alc_get_macaddr(struct alc_softc *sc)
 {
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		alc_get_macaddr_816x(sc);
 	else
 		alc_get_macaddr_813x(sc);
 }
 
 static void
 alc_get_macaddr_813x(struct alc_softc *sc)
 {
 	uint32_t opt;
 	uint16_t val;
 	int eeprom, i;
 
 	eeprom = 0;
 	opt = CSR_READ_4(sc, ALC_OPT_CFG);
 	if ((CSR_READ_4(sc, ALC_MASTER_CFG) & MASTER_OTP_SEL) != 0 &&
 	    (CSR_READ_4(sc, ALC_TWSI_DEBUG) & TWSI_DEBUG_DEV_EXIST) != 0) {
 		/*
 		 * EEPROM found, let TWSI reload EEPROM configuration.
 		 * This will set ethernet address of controller.
 		 */
 		eeprom++;
 		switch (sc->alc_ident->deviceid) {
 		case DEVICEID_ATHEROS_AR8131:
 		case DEVICEID_ATHEROS_AR8132:
 			if ((opt & OPT_CFG_CLK_ENB) == 0) {
 				opt |= OPT_CFG_CLK_ENB;
 				CSR_WRITE_4(sc, ALC_OPT_CFG, opt);
 				CSR_READ_4(sc, ALC_OPT_CFG);
 				DELAY(1000);
 			}
 			break;
 		case DEVICEID_ATHEROS_AR8151:
 		case DEVICEID_ATHEROS_AR8151_V2:
 		case DEVICEID_ATHEROS_AR8152_B:
 		case DEVICEID_ATHEROS_AR8152_B2:
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_ADDR, 0x00);
 			val = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA);
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA, val & 0xFF7F);
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_ADDR, 0x3B);
 			val = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA);
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA, val | 0x0008);
 			DELAY(20);
 			break;
 		}
 
 		CSR_WRITE_4(sc, ALC_LTSSM_ID_CFG,
 		    CSR_READ_4(sc, ALC_LTSSM_ID_CFG) & ~LTSSM_ID_WRO_ENB);
 		CSR_WRITE_4(sc, ALC_WOL_CFG, 0);
 		CSR_READ_4(sc, ALC_WOL_CFG);
 
 		CSR_WRITE_4(sc, ALC_TWSI_CFG, CSR_READ_4(sc, ALC_TWSI_CFG) |
 		    TWSI_CFG_SW_LD_START);
 		for (i = 100; i > 0; i--) {
 			DELAY(1000);
 			if ((CSR_READ_4(sc, ALC_TWSI_CFG) &
 			    TWSI_CFG_SW_LD_START) == 0)
 				break;
 		}
 		if (i == 0)
 			device_printf(sc->alc_dev,
 			    "reloading EEPROM timeout!\n");
 	} else {
 		if (bootverbose)
 			device_printf(sc->alc_dev, "EEPROM not found!\n");
 	}
 	if (eeprom != 0) {
 		switch (sc->alc_ident->deviceid) {
 		case DEVICEID_ATHEROS_AR8131:
 		case DEVICEID_ATHEROS_AR8132:
 			if ((opt & OPT_CFG_CLK_ENB) != 0) {
 				opt &= ~OPT_CFG_CLK_ENB;
 				CSR_WRITE_4(sc, ALC_OPT_CFG, opt);
 				CSR_READ_4(sc, ALC_OPT_CFG);
 				DELAY(1000);
 			}
 			break;
 		case DEVICEID_ATHEROS_AR8151:
 		case DEVICEID_ATHEROS_AR8151_V2:
 		case DEVICEID_ATHEROS_AR8152_B:
 		case DEVICEID_ATHEROS_AR8152_B2:
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_ADDR, 0x00);
 			val = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA);
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA, val | 0x0080);
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_ADDR, 0x3B);
 			val = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA);
 			alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 			    ALC_MII_DBG_DATA, val & 0xFFF7);
 			DELAY(20);
 			break;
 		}
 	}
 
 	alc_get_macaddr_par(sc);
 }
 
 static void
 alc_get_macaddr_816x(struct alc_softc *sc)
 {
 	uint32_t reg;
 	int i, reloaded;
 
 	reloaded = 0;
 	/* Try to reload station address via TWSI. */
 	for (i = 100; i > 0; i--) {
 		reg = CSR_READ_4(sc, ALC_SLD);
 		if ((reg & (SLD_PROGRESS | SLD_START)) == 0)
 			break;
 		DELAY(1000);
 	}
 	if (i != 0) {
 		CSR_WRITE_4(sc, ALC_SLD, reg | SLD_START);
 		for (i = 100; i > 0; i--) {
 			DELAY(1000);
 			reg = CSR_READ_4(sc, ALC_SLD);
 			if ((reg & SLD_START) == 0)
 				break;
 		}
 		if (i != 0)
 			reloaded++;
 		else if (bootverbose)
 			device_printf(sc->alc_dev,
 			    "reloading station address via TWSI timed out!\n");
 	}
 
 	/* Try to reload station address from EEPROM or FLASH. */
 	if (reloaded == 0) {
 		reg = CSR_READ_4(sc, ALC_EEPROM_LD);
 		if ((reg & (EEPROM_LD_EEPROM_EXIST |
 		    EEPROM_LD_FLASH_EXIST)) != 0) {
 			for (i = 100; i > 0; i--) {
 				reg = CSR_READ_4(sc, ALC_EEPROM_LD);
 				if ((reg & (EEPROM_LD_PROGRESS |
 				    EEPROM_LD_START)) == 0)
 					break;
 				DELAY(1000);
 			}
 			if (i != 0) {
 				CSR_WRITE_4(sc, ALC_EEPROM_LD, reg |
 				    EEPROM_LD_START);
 				for (i = 100; i > 0; i--) {
 					DELAY(1000);
 					reg = CSR_READ_4(sc, ALC_EEPROM_LD);
 					if ((reg & EEPROM_LD_START) == 0)
 						break;
 				}
 			} else if (bootverbose)
 				device_printf(sc->alc_dev,
 				    "reloading EEPROM/FLASH timed out!\n");
 		}
 	}
 
 	alc_get_macaddr_par(sc);
 }
 
 static void
 alc_get_macaddr_par(struct alc_softc *sc)
 {
 	uint32_t ea[2];
 
 	ea[0] = CSR_READ_4(sc, ALC_PAR0);
 	ea[1] = CSR_READ_4(sc, ALC_PAR1);
 	sc->alc_eaddr[0] = (ea[1] >> 8) & 0xFF;
 	sc->alc_eaddr[1] = (ea[1] >> 0) & 0xFF;
 	sc->alc_eaddr[2] = (ea[0] >> 24) & 0xFF;
 	sc->alc_eaddr[3] = (ea[0] >> 16) & 0xFF;
 	sc->alc_eaddr[4] = (ea[0] >> 8) & 0xFF;
 	sc->alc_eaddr[5] = (ea[0] >> 0) & 0xFF;
 }
 
 static void
 alc_disable_l0s_l1(struct alc_softc *sc)
 {
 	uint32_t pmcfg;
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		/* Another magic from vendor. */
 		pmcfg = CSR_READ_4(sc, ALC_PM_CFG);
 		pmcfg &= ~(PM_CFG_L1_ENTRY_TIMER_MASK | PM_CFG_CLK_SWH_L1 |
 		    PM_CFG_ASPM_L0S_ENB | PM_CFG_ASPM_L1_ENB |
 		    PM_CFG_MAC_ASPM_CHK | PM_CFG_SERDES_PD_EX_L1);
 		pmcfg |= PM_CFG_SERDES_BUDS_RX_L1_ENB |
 		    PM_CFG_SERDES_PLL_L1_ENB | PM_CFG_SERDES_L1_ENB;
 		CSR_WRITE_4(sc, ALC_PM_CFG, pmcfg);
 	}
 }
 
 static void
 alc_phy_reset(struct alc_softc *sc)
 {
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		alc_phy_reset_816x(sc);
 	else
 		alc_phy_reset_813x(sc);
 }
 
 static void
 alc_phy_reset_813x(struct alc_softc *sc)
 {
 	uint16_t data;
 
 	/* Reset magic from Linux. */
 	CSR_WRITE_2(sc, ALC_GPHY_CFG, GPHY_CFG_SEL_ANA_RESET);
 	CSR_READ_2(sc, ALC_GPHY_CFG);
 	DELAY(10 * 1000);
 
 	CSR_WRITE_2(sc, ALC_GPHY_CFG, GPHY_CFG_EXT_RESET |
 	    GPHY_CFG_SEL_ANA_RESET);
 	CSR_READ_2(sc, ALC_GPHY_CFG);
 	DELAY(10 * 1000);
 
 	/* DSP fixup, Vendor magic. */
 	if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B) {
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_ADDR, 0x000A);
 		data = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_DATA);
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_DATA, data & 0xDFFF);
 	}
 	if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151_V2 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B2) {
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_ADDR, 0x003B);
 		data = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_DATA);
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_DATA, data & 0xFFF7);
 		DELAY(20 * 1000);
 	}
 	if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151) {
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_ADDR, 0x0029);
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_DATA, 0x929D);
 	}
 	if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8131 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8132 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151_V2 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B2) {
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_ADDR, 0x0029);
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    ALC_MII_DBG_DATA, 0xB6DD);
 	}
 
 	/* Load DSP codes, vendor magic. */
 	data = ANA_LOOP_SEL_10BT | ANA_EN_MASK_TB | ANA_EN_10BT_IDLE |
 	    ((1 << ANA_INTERVAL_SEL_TIMER_SHIFT) & ANA_INTERVAL_SEL_TIMER_MASK);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_ADDR, MII_ANA_CFG18);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA, data);
 
 	data = ((2 << ANA_SERDES_CDR_BW_SHIFT) & ANA_SERDES_CDR_BW_MASK) |
 	    ANA_SERDES_EN_DEEM | ANA_SERDES_SEL_HSP | ANA_SERDES_EN_PLL |
 	    ANA_SERDES_EN_LCKDT;
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_ADDR, MII_ANA_CFG5);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA, data);
 
 	data = ((44 << ANA_LONG_CABLE_TH_100_SHIFT) &
 	    ANA_LONG_CABLE_TH_100_MASK) |
 	    ((33 << ANA_SHORT_CABLE_TH_100_SHIFT) &
 	    ANA_SHORT_CABLE_TH_100_SHIFT) |
 	    ANA_BP_BAD_LINK_ACCUM | ANA_BP_SMALL_BW;
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_ADDR, MII_ANA_CFG54);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA, data);
 
 	data = ((11 << ANA_IECHO_ADJ_3_SHIFT) & ANA_IECHO_ADJ_3_MASK) |
 	    ((11 << ANA_IECHO_ADJ_2_SHIFT) & ANA_IECHO_ADJ_2_MASK) |
 	    ((8 << ANA_IECHO_ADJ_1_SHIFT) & ANA_IECHO_ADJ_1_MASK) |
 	    ((8 << ANA_IECHO_ADJ_0_SHIFT) & ANA_IECHO_ADJ_0_MASK);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_ADDR, MII_ANA_CFG4);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA, data);
 
 	data = ((7 & ANA_MANUL_SWICH_ON_SHIFT) & ANA_MANUL_SWICH_ON_MASK) |
 	    ANA_RESTART_CAL | ANA_MAN_ENABLE | ANA_SEL_HSP | ANA_EN_HB |
 	    ANA_OEN_125M;
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_ADDR, MII_ANA_CFG0);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA, data);
 	DELAY(1000);
 
 	/* Disable hibernation. */
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, ALC_MII_DBG_ADDR,
 	    0x0029);
 	data = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA);
 	data &= ~0x8000;
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, ALC_MII_DBG_DATA,
 	    data);
 
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, ALC_MII_DBG_ADDR,
 	    0x000B);
 	data = alc_miibus_readreg(sc->alc_dev, sc->alc_phyaddr,
 	    ALC_MII_DBG_DATA);
 	data &= ~0x8000;
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, ALC_MII_DBG_DATA,
 	    data);
 }
 
 static void
 alc_phy_reset_816x(struct alc_softc *sc)
 {
 	uint32_t val;
 
 	val = CSR_READ_4(sc, ALC_GPHY_CFG);
 	val &= ~(GPHY_CFG_EXT_RESET | GPHY_CFG_LED_MODE |
 	    GPHY_CFG_GATE_25M_ENB | GPHY_CFG_PHY_IDDQ | GPHY_CFG_PHY_PLL_ON |
 	    GPHY_CFG_PWDOWN_HW | GPHY_CFG_100AB_ENB);
 	val |= GPHY_CFG_SEL_ANA_RESET;
 #ifdef notyet
 	val |= GPHY_CFG_HIB_PULSE | GPHY_CFG_HIB_EN | GPHY_CFG_SEL_ANA_RESET;
 #else
 	/* Disable PHY hibernation. */
 	val &= ~(GPHY_CFG_HIB_PULSE | GPHY_CFG_HIB_EN);
 #endif
 	CSR_WRITE_4(sc, ALC_GPHY_CFG, val);
 	DELAY(10);
 	CSR_WRITE_4(sc, ALC_GPHY_CFG, val | GPHY_CFG_EXT_RESET);
 	DELAY(800);
 
 	/* Vendor PHY magic. */
 #ifdef notyet
 	alc_miidbg_writereg(sc, MII_DBG_LEGCYPS, DBG_LEGCYPS_DEFAULT);
 	alc_miidbg_writereg(sc, MII_DBG_SYSMODCTL, DBG_SYSMODCTL_DEFAULT);
 	alc_miiext_writereg(sc, MII_EXT_PCS, MII_EXT_VDRVBIAS,
 	    EXT_VDRVBIAS_DEFAULT);
 #else
 	/* Disable PHY hibernation. */
 	alc_miidbg_writereg(sc, MII_DBG_LEGCYPS,
 	    DBG_LEGCYPS_DEFAULT & ~DBG_LEGCYPS_ENB);
 	alc_miidbg_writereg(sc, MII_DBG_HIBNEG,
 	    DBG_HIBNEG_DEFAULT & ~(DBG_HIBNEG_PSHIB_EN | DBG_HIBNEG_HIB_PULSE));
 	alc_miidbg_writereg(sc, MII_DBG_GREENCFG, DBG_GREENCFG_DEFAULT);
 #endif
 
 	/* XXX Disable EEE. */
 	val = CSR_READ_4(sc, ALC_LPI_CTL);
 	val &= ~LPI_CTL_ENB;
 	CSR_WRITE_4(sc, ALC_LPI_CTL, val);
 	alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_LOCAL_EEEADV, 0);
 
 	/* PHY power saving. */
 	alc_miidbg_writereg(sc, MII_DBG_TST10BTCFG, DBG_TST10BTCFG_DEFAULT);
 	alc_miidbg_writereg(sc, MII_DBG_SRDSYSMOD, DBG_SRDSYSMOD_DEFAULT);
 	alc_miidbg_writereg(sc, MII_DBG_TST100BTCFG, DBG_TST100BTCFG_DEFAULT);
 	alc_miidbg_writereg(sc, MII_DBG_ANACTL, DBG_ANACTL_DEFAULT);
 	val = alc_miidbg_readreg(sc, MII_DBG_GREENCFG2);
 	val &= ~DBG_GREENCFG2_GATE_DFSE_EN;
 	alc_miidbg_writereg(sc, MII_DBG_GREENCFG2, val);
 
 	/* RTL8139C, 120m issue. */
 	alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_NLP78,
 	    ANEG_NLP78_120M_DEFAULT);
 	alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_S3DIG10,
 	    ANEG_S3DIG10_DEFAULT);
 
 	if ((sc->alc_flags & ALC_FLAG_LINK_WAR) != 0) {
 		/* Turn off half amplitude. */
 		val = alc_miiext_readreg(sc, MII_EXT_PCS, MII_EXT_CLDCTL3);
 		val |= EXT_CLDCTL3_BP_CABLE1TH_DET_GT;
 		alc_miiext_writereg(sc, MII_EXT_PCS, MII_EXT_CLDCTL3, val);
 		/* Turn off Green feature. */
 		val = alc_miidbg_readreg(sc, MII_DBG_GREENCFG2);
 		val |= DBG_GREENCFG2_BP_GREEN;
 		alc_miidbg_writereg(sc, MII_DBG_GREENCFG2, val);
 		/* Turn off half bias. */
 		val = alc_miiext_readreg(sc, MII_EXT_PCS, MII_EXT_CLDCTL5);
 		val |= EXT_CLDCTL5_BP_VD_HLFBIAS;
 		alc_miiext_writereg(sc, MII_EXT_PCS, MII_EXT_CLDCTL5, val);
 	}
 }
 
 static void
 alc_phy_down(struct alc_softc *sc)
 {
 	uint32_t gphy;
 
 	switch (sc->alc_ident->deviceid) {
 	case DEVICEID_ATHEROS_AR8161:
 	case DEVICEID_ATHEROS_E2200:
 	case DEVICEID_ATHEROS_AR8162:
 	case DEVICEID_ATHEROS_AR8171:
 	case DEVICEID_ATHEROS_AR8172:
 		gphy = CSR_READ_4(sc, ALC_GPHY_CFG);
 		gphy &= ~(GPHY_CFG_EXT_RESET | GPHY_CFG_LED_MODE |
 		    GPHY_CFG_100AB_ENB | GPHY_CFG_PHY_PLL_ON);
 		gphy |= GPHY_CFG_HIB_EN | GPHY_CFG_HIB_PULSE |
 		    GPHY_CFG_SEL_ANA_RESET;
 		gphy |= GPHY_CFG_PHY_IDDQ | GPHY_CFG_PWDOWN_HW;
 		CSR_WRITE_4(sc, ALC_GPHY_CFG, gphy);
 		break;
 	case DEVICEID_ATHEROS_AR8151:
 	case DEVICEID_ATHEROS_AR8151_V2:
 	case DEVICEID_ATHEROS_AR8152_B:
 	case DEVICEID_ATHEROS_AR8152_B2:
 		/*
 		 * GPHY power down caused more problems on AR8151 v2.0.
 		 * When driver is reloaded after GPHY power down,
 		 * accesses to PHY/MAC registers hung the system. Only
 		 * cold boot recovered from it.  I'm not sure whether
 		 * AR8151 v1.0 also requires this one though.  I don't
 		 * have AR8151 v1.0 controller in hand.
 		 * The only option left is to isolate the PHY and
 		 * initiates power down the PHY which in turn saves
 		 * more power when driver is unloaded.
 		 */
 		alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 		    MII_BMCR, BMCR_ISO | BMCR_PDOWN);
 		break;
 	default:
 		/* Force PHY down. */
 		CSR_WRITE_2(sc, ALC_GPHY_CFG, GPHY_CFG_EXT_RESET |
 		    GPHY_CFG_SEL_ANA_RESET | GPHY_CFG_PHY_IDDQ |
 		    GPHY_CFG_PWDOWN_HW);
 		DELAY(1000);
 		break;
 	}
 }
 
 static void
 alc_aspm(struct alc_softc *sc, int init, int media)
 {
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		alc_aspm_816x(sc, init);
 	else
 		alc_aspm_813x(sc, media);
 }
 
 static void
 alc_aspm_813x(struct alc_softc *sc, int media)
 {
 	uint32_t pmcfg;
 	uint16_t linkcfg;
 
 	if ((sc->alc_flags & ALC_FLAG_LINK) == 0)
 		return;
 
 	pmcfg = CSR_READ_4(sc, ALC_PM_CFG);
 	if ((sc->alc_flags & (ALC_FLAG_APS | ALC_FLAG_PCIE)) ==
 	    (ALC_FLAG_APS | ALC_FLAG_PCIE))
 		linkcfg = CSR_READ_2(sc, sc->alc_expcap +
 		    PCIER_LINK_CTL);
 	else
 		linkcfg = 0;
 	pmcfg &= ~PM_CFG_SERDES_PD_EX_L1;
 	pmcfg &= ~(PM_CFG_L1_ENTRY_TIMER_MASK | PM_CFG_LCKDET_TIMER_MASK);
 	pmcfg |= PM_CFG_MAC_ASPM_CHK;
 	pmcfg |= (PM_CFG_LCKDET_TIMER_DEFAULT << PM_CFG_LCKDET_TIMER_SHIFT);
 	pmcfg &= ~(PM_CFG_ASPM_L1_ENB | PM_CFG_ASPM_L0S_ENB);
 
 	if ((sc->alc_flags & ALC_FLAG_APS) != 0) {
 		/* Disable extended sync except AR8152 B v1.0 */
 		linkcfg &= ~PCIEM_LINK_CTL_EXTENDED_SYNC;
 		if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B &&
 		    sc->alc_rev == ATHEROS_AR8152_B_V10)
 			linkcfg |= PCIEM_LINK_CTL_EXTENDED_SYNC;
 		CSR_WRITE_2(sc, sc->alc_expcap + PCIER_LINK_CTL,
 		    linkcfg);
 		pmcfg &= ~(PM_CFG_EN_BUFS_RX_L0S | PM_CFG_SA_DLY_ENB |
 		    PM_CFG_HOTRST);
 		pmcfg |= (PM_CFG_L1_ENTRY_TIMER_DEFAULT <<
 		    PM_CFG_L1_ENTRY_TIMER_SHIFT);
 		pmcfg &= ~PM_CFG_PM_REQ_TIMER_MASK;
 		pmcfg |= (PM_CFG_PM_REQ_TIMER_DEFAULT <<
 		    PM_CFG_PM_REQ_TIMER_SHIFT);
 		pmcfg |= PM_CFG_SERDES_PD_EX_L1 | PM_CFG_PCIE_RECV;
 	}
 
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0) {
 		if ((sc->alc_flags & ALC_FLAG_L0S) != 0)
 			pmcfg |= PM_CFG_ASPM_L0S_ENB;
 		if ((sc->alc_flags & ALC_FLAG_L1S) != 0)
 			pmcfg |= PM_CFG_ASPM_L1_ENB;
 		if ((sc->alc_flags & ALC_FLAG_APS) != 0) {
 			if (sc->alc_ident->deviceid ==
 			    DEVICEID_ATHEROS_AR8152_B)
 				pmcfg &= ~PM_CFG_ASPM_L0S_ENB;
 			pmcfg &= ~(PM_CFG_SERDES_L1_ENB |
 			    PM_CFG_SERDES_PLL_L1_ENB |
 			    PM_CFG_SERDES_BUDS_RX_L1_ENB);
 			pmcfg |= PM_CFG_CLK_SWH_L1;
 			if (media == IFM_100_TX || media == IFM_1000_T) {
 				pmcfg &= ~PM_CFG_L1_ENTRY_TIMER_MASK;
 				switch (sc->alc_ident->deviceid) {
 				case DEVICEID_ATHEROS_AR8152_B:
 					pmcfg |= (7 <<
 					    PM_CFG_L1_ENTRY_TIMER_SHIFT);
 					break;
 				case DEVICEID_ATHEROS_AR8152_B2:
 				case DEVICEID_ATHEROS_AR8151_V2:
 					pmcfg |= (4 <<
 					    PM_CFG_L1_ENTRY_TIMER_SHIFT);
 					break;
 				default:
 					pmcfg |= (15 <<
 					    PM_CFG_L1_ENTRY_TIMER_SHIFT);
 					break;
 				}
 			}
 		} else {
 			pmcfg |= PM_CFG_SERDES_L1_ENB |
 			    PM_CFG_SERDES_PLL_L1_ENB |
 			    PM_CFG_SERDES_BUDS_RX_L1_ENB;
 			pmcfg &= ~(PM_CFG_CLK_SWH_L1 |
 			    PM_CFG_ASPM_L1_ENB | PM_CFG_ASPM_L0S_ENB);
 		}
 	} else {
 		pmcfg &= ~(PM_CFG_SERDES_BUDS_RX_L1_ENB | PM_CFG_SERDES_L1_ENB |
 		    PM_CFG_SERDES_PLL_L1_ENB);
 		pmcfg |= PM_CFG_CLK_SWH_L1;
 		if ((sc->alc_flags & ALC_FLAG_L1S) != 0)
 			pmcfg |= PM_CFG_ASPM_L1_ENB;
 	}
 	CSR_WRITE_4(sc, ALC_PM_CFG, pmcfg);
 }
 
 static void
 alc_aspm_816x(struct alc_softc *sc, int init)
 {
 	uint32_t pmcfg;
 
 	pmcfg = CSR_READ_4(sc, ALC_PM_CFG);
 	pmcfg &= ~PM_CFG_L1_ENTRY_TIMER_816X_MASK;
 	pmcfg |= PM_CFG_L1_ENTRY_TIMER_816X_DEFAULT;
 	pmcfg &= ~PM_CFG_PM_REQ_TIMER_MASK;
 	pmcfg |= PM_CFG_PM_REQ_TIMER_816X_DEFAULT;
 	pmcfg &= ~PM_CFG_LCKDET_TIMER_MASK;
 	pmcfg |= PM_CFG_LCKDET_TIMER_DEFAULT;
 	pmcfg |= PM_CFG_SERDES_PD_EX_L1 | PM_CFG_CLK_SWH_L1 | PM_CFG_PCIE_RECV;
 	pmcfg &= ~(PM_CFG_RX_L1_AFTER_L0S | PM_CFG_TX_L1_AFTER_L0S |
 	    PM_CFG_ASPM_L1_ENB | PM_CFG_ASPM_L0S_ENB |
 	    PM_CFG_SERDES_L1_ENB | PM_CFG_SERDES_PLL_L1_ENB |
 	    PM_CFG_SERDES_BUDS_RX_L1_ENB | PM_CFG_SA_DLY_ENB |
 	    PM_CFG_MAC_ASPM_CHK | PM_CFG_HOTRST);
 	if (AR816X_REV(sc->alc_rev) <= AR816X_REV_A1 &&
 	    (sc->alc_rev & 0x01) != 0)
 		pmcfg |= PM_CFG_SERDES_L1_ENB | PM_CFG_SERDES_PLL_L1_ENB;
 	if ((sc->alc_flags & ALC_FLAG_LINK) != 0) {
 		/* Link up, enable both L0s, L1s. */
 		pmcfg |= PM_CFG_ASPM_L0S_ENB | PM_CFG_ASPM_L1_ENB |
 		    PM_CFG_MAC_ASPM_CHK;
 	} else {
 		if (init != 0)
 			pmcfg |= PM_CFG_ASPM_L0S_ENB | PM_CFG_ASPM_L1_ENB |
 			    PM_CFG_MAC_ASPM_CHK;
 		else if ((sc->alc_ifp->if_drv_flags & IFF_DRV_RUNNING) != 0)
 			pmcfg |= PM_CFG_ASPM_L1_ENB | PM_CFG_MAC_ASPM_CHK;
 	}
 	CSR_WRITE_4(sc, ALC_PM_CFG, pmcfg);
 }
 
 static void
 alc_init_pcie(struct alc_softc *sc)
 {
 	const char *aspm_state[] = { "L0s/L1", "L0s", "L1", "L0s/L1" };
 	uint32_t cap, ctl, val;
 	int state;
 
 	/* Clear data link and flow-control protocol error. */
 	val = CSR_READ_4(sc, ALC_PEX_UNC_ERR_SEV);
 	val &= ~(PEX_UNC_ERR_SEV_DLP | PEX_UNC_ERR_SEV_FCP);
 	CSR_WRITE_4(sc, ALC_PEX_UNC_ERR_SEV, val);
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		CSR_WRITE_4(sc, ALC_LTSSM_ID_CFG,
 		    CSR_READ_4(sc, ALC_LTSSM_ID_CFG) & ~LTSSM_ID_WRO_ENB);
 		CSR_WRITE_4(sc, ALC_PCIE_PHYMISC,
 		    CSR_READ_4(sc, ALC_PCIE_PHYMISC) |
 		    PCIE_PHYMISC_FORCE_RCV_DET);
 		if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B &&
 		    sc->alc_rev == ATHEROS_AR8152_B_V10) {
 			val = CSR_READ_4(sc, ALC_PCIE_PHYMISC2);
 			val &= ~(PCIE_PHYMISC2_SERDES_CDR_MASK |
 			    PCIE_PHYMISC2_SERDES_TH_MASK);
 			val |= 3 << PCIE_PHYMISC2_SERDES_CDR_SHIFT;
 			val |= 3 << PCIE_PHYMISC2_SERDES_TH_SHIFT;
 			CSR_WRITE_4(sc, ALC_PCIE_PHYMISC2, val);
 		}
 		/* Disable ASPM L0S and L1. */
 		cap = CSR_READ_2(sc, sc->alc_expcap + PCIER_LINK_CAP);
 		if ((cap & PCIEM_LINK_CAP_ASPM) != 0) {
 			ctl = CSR_READ_2(sc, sc->alc_expcap + PCIER_LINK_CTL);
 			if ((ctl & PCIEM_LINK_CTL_RCB) != 0)
 				sc->alc_rcb = DMA_CFG_RCB_128;
 			if (bootverbose)
 				device_printf(sc->alc_dev, "RCB %u bytes\n",
 				    sc->alc_rcb == DMA_CFG_RCB_64 ? 64 : 128);
 			state = ctl & PCIEM_LINK_CTL_ASPMC;
 			if (state & PCIEM_LINK_CTL_ASPMC_L0S)
 				sc->alc_flags |= ALC_FLAG_L0S;
 			if (state & PCIEM_LINK_CTL_ASPMC_L1)
 				sc->alc_flags |= ALC_FLAG_L1S;
 			if (bootverbose)
 				device_printf(sc->alc_dev, "ASPM %s %s\n",
 				    aspm_state[state],
 				    state == 0 ? "disabled" : "enabled");
 			alc_disable_l0s_l1(sc);
 		} else {
 			if (bootverbose)
 				device_printf(sc->alc_dev,
 				    "no ASPM support\n");
 		}
 	} else {
 		val = CSR_READ_4(sc, ALC_PDLL_TRNS1);
 		val &= ~PDLL_TRNS1_D3PLLOFF_ENB;
 		CSR_WRITE_4(sc, ALC_PDLL_TRNS1, val);
 		val = CSR_READ_4(sc, ALC_MASTER_CFG);
 		if (AR816X_REV(sc->alc_rev) <= AR816X_REV_A1 &&
 		    (sc->alc_rev & 0x01) != 0) {
 			if ((val & MASTER_WAKEN_25M) == 0 ||
 			    (val & MASTER_CLK_SEL_DIS) == 0) {
 				val |= MASTER_WAKEN_25M | MASTER_CLK_SEL_DIS;
 				CSR_WRITE_4(sc, ALC_MASTER_CFG, val);
 			}
 		} else {
 			if ((val & MASTER_WAKEN_25M) == 0 ||
 			    (val & MASTER_CLK_SEL_DIS) != 0) {
 				val |= MASTER_WAKEN_25M;
 				val &= ~MASTER_CLK_SEL_DIS;
 				CSR_WRITE_4(sc, ALC_MASTER_CFG, val);
 			}
 		}
 	}
 	alc_aspm(sc, 1, IFM_UNKNOWN);
 }
 
 static void
 alc_config_msi(struct alc_softc *sc)
 {
 	uint32_t ctl, mod;
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		/*
 		 * It seems interrupt moderation is controlled by
 		 * ALC_MSI_RETRANS_TIMER register if MSI/MSIX is active.
 		 * Driver uses RX interrupt moderation parameter to
 		 * program ALC_MSI_RETRANS_TIMER register.
 		 */
 		ctl = CSR_READ_4(sc, ALC_MSI_RETRANS_TIMER);
 		ctl &= ~MSI_RETRANS_TIMER_MASK;
 		ctl &= ~MSI_RETRANS_MASK_SEL_LINE;
 		mod = ALC_USECS(sc->alc_int_rx_mod);
 		if (mod == 0)
 			mod = 1;
 		ctl |= mod;
 		if ((sc->alc_flags & ALC_FLAG_MSIX) != 0)
 			CSR_WRITE_4(sc, ALC_MSI_RETRANS_TIMER, ctl |
 			    MSI_RETRANS_MASK_SEL_STD);
 		else if ((sc->alc_flags & ALC_FLAG_MSI) != 0)
 			CSR_WRITE_4(sc, ALC_MSI_RETRANS_TIMER, ctl |
 			    MSI_RETRANS_MASK_SEL_LINE);
 		else
 			CSR_WRITE_4(sc, ALC_MSI_RETRANS_TIMER, 0);
 	}
 }
 
 static int
 alc_attach(device_t dev)
 {
 	struct alc_softc *sc;
 	struct ifnet *ifp;
 	int base, error, i, msic, msixc;
 	uint16_t burst;
 
 	error = 0;
 	sc = device_get_softc(dev);
 	sc->alc_dev = dev;
 	sc->alc_rev = pci_get_revid(dev);
 
 	mtx_init(&sc->alc_mtx, device_get_nameunit(dev), MTX_NETWORK_LOCK,
 	    MTX_DEF);
 	callout_init_mtx(&sc->alc_tick_ch, &sc->alc_mtx, 0);
 	TASK_INIT(&sc->alc_int_task, 0, alc_int_task, sc);
 	sc->alc_ident = alc_find_ident(dev);
 
 	/* Map the device. */
 	pci_enable_busmaster(dev);
 	sc->alc_res_spec = alc_res_spec_mem;
 	sc->alc_irq_spec = alc_irq_spec_legacy;
 	error = bus_alloc_resources(dev, sc->alc_res_spec, sc->alc_res);
 	if (error != 0) {
 		device_printf(dev, "cannot allocate memory resources.\n");
 		goto fail;
 	}
 
 	/* Set PHY address. */
 	sc->alc_phyaddr = ALC_PHY_ADDR;
 
 	/*
 	 * One odd thing is AR8132 uses the same PHY hardware(F1
 	 * gigabit PHY) of AR8131. So atphy(4) of AR8132 reports
 	 * the PHY supports 1000Mbps but that's not true. The PHY
 	 * used in AR8132 can't establish gigabit link even if it
 	 * shows the same PHY model/revision number of AR8131.
 	 */
 	switch (sc->alc_ident->deviceid) {
 	case DEVICEID_ATHEROS_AR8161:
 		if (pci_get_subvendor(dev) == VENDORID_ATHEROS &&
 		    pci_get_subdevice(dev) == 0x0091 && sc->alc_rev == 0)
 			sc->alc_flags |= ALC_FLAG_LINK_WAR;
 		/* FALLTHROUGH */
 	case DEVICEID_ATHEROS_E2200:
 	case DEVICEID_ATHEROS_AR8171:
 		sc->alc_flags |= ALC_FLAG_AR816X_FAMILY;
 		break;
 	case DEVICEID_ATHEROS_AR8162:
 	case DEVICEID_ATHEROS_AR8172:
 		sc->alc_flags |= ALC_FLAG_FASTETHER | ALC_FLAG_AR816X_FAMILY;
 		break;
 	case DEVICEID_ATHEROS_AR8152_B:
 	case DEVICEID_ATHEROS_AR8152_B2:
 		sc->alc_flags |= ALC_FLAG_APS;
 		/* FALLTHROUGH */
 	case DEVICEID_ATHEROS_AR8132:
 		sc->alc_flags |= ALC_FLAG_FASTETHER;
 		break;
 	case DEVICEID_ATHEROS_AR8151:
 	case DEVICEID_ATHEROS_AR8151_V2:
 		sc->alc_flags |= ALC_FLAG_APS;
 		/* FALLTHROUGH */
 	default:
 		break;
 	}
 	sc->alc_flags |= ALC_FLAG_JUMBO;
 
 	/*
 	 * It seems that AR813x/AR815x has silicon bug for SMB. In
 	 * addition, Atheros said that enabling SMB wouldn't improve
 	 * performance. However I think it's bad to access lots of
 	 * registers to extract MAC statistics.
 	 */
 	sc->alc_flags |= ALC_FLAG_SMB_BUG;
 	/*
 	 * Don't use Tx CMB. It is known to have silicon bug.
 	 */
 	sc->alc_flags |= ALC_FLAG_CMB_BUG;
 	sc->alc_chip_rev = CSR_READ_4(sc, ALC_MASTER_CFG) >>
 	    MASTER_CHIP_REV_SHIFT;
 	if (bootverbose) {
 		device_printf(dev, "PCI device revision : 0x%04x\n",
 		    sc->alc_rev);
 		device_printf(dev, "Chip id/revision : 0x%04x\n",
 		    sc->alc_chip_rev);
 		if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 			device_printf(dev, "AR816x revision : 0x%x\n",
 			    AR816X_REV(sc->alc_rev));
 	}
 	device_printf(dev, "%u Tx FIFO, %u Rx FIFO\n",
 	    CSR_READ_4(sc, ALC_SRAM_TX_FIFO_LEN) * 8,
 	    CSR_READ_4(sc, ALC_SRAM_RX_FIFO_LEN) * 8);
 
 	/* Initialize DMA parameters. */
 	sc->alc_dma_rd_burst = 0;
 	sc->alc_dma_wr_burst = 0;
 	sc->alc_rcb = DMA_CFG_RCB_64;
 	if (pci_find_cap(dev, PCIY_EXPRESS, &base) == 0) {
 		sc->alc_flags |= ALC_FLAG_PCIE;
 		sc->alc_expcap = base;
 		burst = CSR_READ_2(sc, base + PCIER_DEVICE_CTL);
 		sc->alc_dma_rd_burst =
 		    (burst & PCIEM_CTL_MAX_READ_REQUEST) >> 12;
 		sc->alc_dma_wr_burst = (burst & PCIEM_CTL_MAX_PAYLOAD) >> 5;
 		if (bootverbose) {
 			device_printf(dev, "Read request size : %u bytes.\n",
 			    alc_dma_burst[sc->alc_dma_rd_burst]);
 			device_printf(dev, "TLP payload size : %u bytes.\n",
 			    alc_dma_burst[sc->alc_dma_wr_burst]);
 		}
 		if (alc_dma_burst[sc->alc_dma_rd_burst] > 1024)
 			sc->alc_dma_rd_burst = 3;
 		if (alc_dma_burst[sc->alc_dma_wr_burst] > 1024)
 			sc->alc_dma_wr_burst = 3;
 		alc_init_pcie(sc);
 	}
 
 	/* Reset PHY. */
 	alc_phy_reset(sc);
 
 	/* Reset the ethernet controller. */
 	alc_stop_mac(sc);
 	alc_reset(sc);
 
 	/* Allocate IRQ resources. */
 	msixc = pci_msix_count(dev);
 	msic = pci_msi_count(dev);
 	if (bootverbose) {
 		device_printf(dev, "MSIX count : %d\n", msixc);
 		device_printf(dev, "MSI count : %d\n", msic);
 	}
 	if (msixc > 1)
 		msixc = 1;
 	if (msic > 1)
 		msic = 1;
 	/*
 	 * Prefer MSIX over MSI.
 	 * AR816x controller has a silicon bug that MSI interrupt
 	 * does not assert if PCIM_CMD_INTxDIS bit of command
 	 * register is set.  pci(4) was taught to handle that case.
 	 */
 	if (msix_disable == 0 || msi_disable == 0) {
 		if (msix_disable == 0 && msixc > 0 &&
 		    pci_alloc_msix(dev, &msixc) == 0) {
 			if (msic == 1) {
 				device_printf(dev,
 				    "Using %d MSIX message(s).\n", msixc);
 				sc->alc_flags |= ALC_FLAG_MSIX;
 				sc->alc_irq_spec = alc_irq_spec_msix;
 			} else
 				pci_release_msi(dev);
 		}
 		if (msi_disable == 0 && (sc->alc_flags & ALC_FLAG_MSIX) == 0 &&
 		    msic > 0 && pci_alloc_msi(dev, &msic) == 0) {
 			if (msic == 1) {
 				device_printf(dev,
 				    "Using %d MSI message(s).\n", msic);
 				sc->alc_flags |= ALC_FLAG_MSI;
 				sc->alc_irq_spec = alc_irq_spec_msi;
 			} else
 				pci_release_msi(dev);
 		}
 	}
 
 	error = bus_alloc_resources(dev, sc->alc_irq_spec, sc->alc_irq);
 	if (error != 0) {
 		device_printf(dev, "cannot allocate IRQ resources.\n");
 		goto fail;
 	}
 
 	/* Create device sysctl node. */
 	alc_sysctl_node(sc);
 
 	if ((error = alc_dma_alloc(sc) != 0))
 		goto fail;
 
 	/* Load station address. */
 	alc_get_macaddr(sc);
 
 	ifp = sc->alc_ifp = if_alloc(IFT_ETHER);
 	if (ifp == NULL) {
 		device_printf(dev, "cannot allocate ifnet structure.\n");
 		error = ENXIO;
 		goto fail;
 	}
 
 	ifp->if_softc = sc;
 	if_initname(ifp, device_get_name(dev), device_get_unit(dev));
 	ifp->if_flags = IFF_BROADCAST | IFF_SIMPLEX | IFF_MULTICAST;
 	ifp->if_ioctl = alc_ioctl;
 	ifp->if_start = alc_start;
 	ifp->if_init = alc_init;
 	ifp->if_snd.ifq_drv_maxlen = ALC_TX_RING_CNT - 1;
 	IFQ_SET_MAXLEN(&ifp->if_snd, ifp->if_snd.ifq_drv_maxlen);
 	IFQ_SET_READY(&ifp->if_snd);
 	ifp->if_capabilities = IFCAP_TXCSUM | IFCAP_TSO4;
 	ifp->if_hwassist = ALC_CSUM_FEATURES | CSUM_TSO;
 	if (pci_find_cap(dev, PCIY_PMG, &base) == 0) {
 		ifp->if_capabilities |= IFCAP_WOL_MAGIC | IFCAP_WOL_MCAST;
 		sc->alc_flags |= ALC_FLAG_PM;
 		sc->alc_pmcap = base;
 	}
 	ifp->if_capenable = ifp->if_capabilities;
 
 	/* Set up MII bus. */
 	error = mii_attach(dev, &sc->alc_miibus, ifp, alc_mediachange,
 	    alc_mediastatus, BMSR_DEFCAPMASK, sc->alc_phyaddr, MII_OFFSET_ANY,
 	    MIIF_DOPAUSE);
 	if (error != 0) {
 		device_printf(dev, "attaching PHYs failed\n");
 		goto fail;
 	}
 
 	ether_ifattach(ifp, sc->alc_eaddr);
 
 	/* VLAN capability setup. */
 	ifp->if_capabilities |= IFCAP_VLAN_MTU | IFCAP_VLAN_HWTAGGING |
 	    IFCAP_VLAN_HWCSUM | IFCAP_VLAN_HWTSO;
 	ifp->if_capenable = ifp->if_capabilities;
 	/*
 	 * XXX
 	 * It seems enabling Tx checksum offloading makes more trouble.
 	 * Sometimes the controller does not receive any frames when
 	 * Tx checksum offloading is enabled. I'm not sure whether this
 	 * is a bug in Tx checksum offloading logic or I got broken
 	 * sample boards. To safety, don't enable Tx checksum offloading
 	 * by default but give chance to users to toggle it if they know
 	 * their controllers work without problems.
 	 * Fortunately, Tx checksum offloading for AR816x family
 	 * seems to work.
 	 */
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		ifp->if_capenable &= ~IFCAP_TXCSUM;
 		ifp->if_hwassist &= ~ALC_CSUM_FEATURES;
 	}
 
 	/* Tell the upper layer(s) we support long frames. */
 	ifp->if_hdrlen = sizeof(struct ether_vlan_header);
 
 	/* Create local taskq. */
 	sc->alc_tq = taskqueue_create_fast("alc_taskq", M_WAITOK,
 	    taskqueue_thread_enqueue, &sc->alc_tq);
 	if (sc->alc_tq == NULL) {
 		device_printf(dev, "could not create taskqueue.\n");
 		ether_ifdetach(ifp);
 		error = ENXIO;
 		goto fail;
 	}
 	taskqueue_start_threads(&sc->alc_tq, 1, PI_NET, "%s taskq",
 	    device_get_nameunit(sc->alc_dev));
 
 	alc_config_msi(sc);
 	if ((sc->alc_flags & ALC_FLAG_MSIX) != 0)
 		msic = ALC_MSIX_MESSAGES;
 	else if ((sc->alc_flags & ALC_FLAG_MSI) != 0)
 		msic = ALC_MSI_MESSAGES;
 	else
 		msic = 1;
 	for (i = 0; i < msic; i++) {
 		error = bus_setup_intr(dev, sc->alc_irq[i],
 		    INTR_TYPE_NET | INTR_MPSAFE, alc_intr, NULL, sc,
 		    &sc->alc_intrhand[i]);
 		if (error != 0)
 			break;
 	}
 	if (error != 0) {
 		device_printf(dev, "could not set up interrupt handler.\n");
 		taskqueue_free(sc->alc_tq);
 		sc->alc_tq = NULL;
 		ether_ifdetach(ifp);
 		goto fail;
 	}
 
 fail:
 	if (error != 0)
 		alc_detach(dev);
 
 	return (error);
 }
 
 static int
 alc_detach(device_t dev)
 {
 	struct alc_softc *sc;
 	struct ifnet *ifp;
 	int i, msic;
 
 	sc = device_get_softc(dev);
 
 	ifp = sc->alc_ifp;
 	if (device_is_attached(dev)) {
 		ether_ifdetach(ifp);
 		ALC_LOCK(sc);
 		alc_stop(sc);
 		ALC_UNLOCK(sc);
 		callout_drain(&sc->alc_tick_ch);
 		taskqueue_drain(sc->alc_tq, &sc->alc_int_task);
 	}
 
 	if (sc->alc_tq != NULL) {
 		taskqueue_drain(sc->alc_tq, &sc->alc_int_task);
 		taskqueue_free(sc->alc_tq);
 		sc->alc_tq = NULL;
 	}
 
 	if (sc->alc_miibus != NULL) {
 		device_delete_child(dev, sc->alc_miibus);
 		sc->alc_miibus = NULL;
 	}
 	bus_generic_detach(dev);
 	alc_dma_free(sc);
 
 	if (ifp != NULL) {
 		if_free(ifp);
 		sc->alc_ifp = NULL;
 	}
 
 	if ((sc->alc_flags & ALC_FLAG_MSIX) != 0)
 		msic = ALC_MSIX_MESSAGES;
 	else if ((sc->alc_flags & ALC_FLAG_MSI) != 0)
 		msic = ALC_MSI_MESSAGES;
 	else
 		msic = 1;
 	for (i = 0; i < msic; i++) {
 		if (sc->alc_intrhand[i] != NULL) {
 			bus_teardown_intr(dev, sc->alc_irq[i],
 			    sc->alc_intrhand[i]);
 			sc->alc_intrhand[i] = NULL;
 		}
 	}
 	if (sc->alc_res[0] != NULL)
 		alc_phy_down(sc);
 	bus_release_resources(dev, sc->alc_irq_spec, sc->alc_irq);
 	if ((sc->alc_flags & (ALC_FLAG_MSI | ALC_FLAG_MSIX)) != 0)
 		pci_release_msi(dev);
 	bus_release_resources(dev, sc->alc_res_spec, sc->alc_res);
 	mtx_destroy(&sc->alc_mtx);
 
 	return (0);
 }
 
 #define	ALC_SYSCTL_STAT_ADD32(c, h, n, p, d)	\
 	    SYSCTL_ADD_UINT(c, h, OID_AUTO, n, CTLFLAG_RD, p, 0, d)
 #define	ALC_SYSCTL_STAT_ADD64(c, h, n, p, d)	\
 	    SYSCTL_ADD_UQUAD(c, h, OID_AUTO, n, CTLFLAG_RD, p, d)
 
 static void
 alc_sysctl_node(struct alc_softc *sc)
 {
 	struct sysctl_ctx_list *ctx;
 	struct sysctl_oid_list *child, *parent;
 	struct sysctl_oid *tree;
 	struct alc_hw_stats *stats;
 	int error;
 
 	stats = &sc->alc_stats;
 	ctx = device_get_sysctl_ctx(sc->alc_dev);
 	child = SYSCTL_CHILDREN(device_get_sysctl_tree(sc->alc_dev));
 
 	SYSCTL_ADD_PROC(ctx, child, OID_AUTO, "int_rx_mod",
 	    CTLTYPE_INT | CTLFLAG_RW, &sc->alc_int_rx_mod, 0,
 	    sysctl_hw_alc_int_mod, "I", "alc Rx interrupt moderation");
 	SYSCTL_ADD_PROC(ctx, child, OID_AUTO, "int_tx_mod",
 	    CTLTYPE_INT | CTLFLAG_RW, &sc->alc_int_tx_mod, 0,
 	    sysctl_hw_alc_int_mod, "I", "alc Tx interrupt moderation");
 	/* Pull in device tunables. */
 	sc->alc_int_rx_mod = ALC_IM_RX_TIMER_DEFAULT;
 	error = resource_int_value(device_get_name(sc->alc_dev),
 	    device_get_unit(sc->alc_dev), "int_rx_mod", &sc->alc_int_rx_mod);
 	if (error == 0) {
 		if (sc->alc_int_rx_mod < ALC_IM_TIMER_MIN ||
 		    sc->alc_int_rx_mod > ALC_IM_TIMER_MAX) {
 			device_printf(sc->alc_dev, "int_rx_mod value out of "
 			    "range; using default: %d\n",
 			    ALC_IM_RX_TIMER_DEFAULT);
 			sc->alc_int_rx_mod = ALC_IM_RX_TIMER_DEFAULT;
 		}
 	}
 	sc->alc_int_tx_mod = ALC_IM_TX_TIMER_DEFAULT;
 	error = resource_int_value(device_get_name(sc->alc_dev),
 	    device_get_unit(sc->alc_dev), "int_tx_mod", &sc->alc_int_tx_mod);
 	if (error == 0) {
 		if (sc->alc_int_tx_mod < ALC_IM_TIMER_MIN ||
 		    sc->alc_int_tx_mod > ALC_IM_TIMER_MAX) {
 			device_printf(sc->alc_dev, "int_tx_mod value out of "
 			    "range; using default: %d\n",
 			    ALC_IM_TX_TIMER_DEFAULT);
 			sc->alc_int_tx_mod = ALC_IM_TX_TIMER_DEFAULT;
 		}
 	}
 	SYSCTL_ADD_PROC(ctx, child, OID_AUTO, "process_limit",
 	    CTLTYPE_INT | CTLFLAG_RW, &sc->alc_process_limit, 0,
 	    sysctl_hw_alc_proc_limit, "I",
 	    "max number of Rx events to process");
 	/* Pull in device tunables. */
 	sc->alc_process_limit = ALC_PROC_DEFAULT;
 	error = resource_int_value(device_get_name(sc->alc_dev),
 	    device_get_unit(sc->alc_dev), "process_limit",
 	    &sc->alc_process_limit);
 	if (error == 0) {
 		if (sc->alc_process_limit < ALC_PROC_MIN ||
 		    sc->alc_process_limit > ALC_PROC_MAX) {
 			device_printf(sc->alc_dev,
 			    "process_limit value out of range; "
 			    "using default: %d\n", ALC_PROC_DEFAULT);
 			sc->alc_process_limit = ALC_PROC_DEFAULT;
 		}
 	}
 
 	tree = SYSCTL_ADD_NODE(ctx, child, OID_AUTO, "stats", CTLFLAG_RD,
 	    NULL, "ALC statistics");
 	parent = SYSCTL_CHILDREN(tree);
 
 	/* Rx statistics. */
 	tree = SYSCTL_ADD_NODE(ctx, parent, OID_AUTO, "rx", CTLFLAG_RD,
 	    NULL, "Rx MAC statistics");
 	child = SYSCTL_CHILDREN(tree);
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "good_frames",
 	    &stats->rx_frames, "Good frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "good_bcast_frames",
 	    &stats->rx_bcast_frames, "Good broadcast frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "good_mcast_frames",
 	    &stats->rx_mcast_frames, "Good multicast frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "pause_frames",
 	    &stats->rx_pause_frames, "Pause control frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "control_frames",
 	    &stats->rx_control_frames, "Control frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "crc_errs",
 	    &stats->rx_crcerrs, "CRC errors");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "len_errs",
 	    &stats->rx_lenerrs, "Frames with length mismatched");
 	ALC_SYSCTL_STAT_ADD64(ctx, child, "good_octets",
 	    &stats->rx_bytes, "Good octets");
 	ALC_SYSCTL_STAT_ADD64(ctx, child, "good_bcast_octets",
 	    &stats->rx_bcast_bytes, "Good broadcast octets");
 	ALC_SYSCTL_STAT_ADD64(ctx, child, "good_mcast_octets",
 	    &stats->rx_mcast_bytes, "Good multicast octets");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "runts",
 	    &stats->rx_runts, "Too short frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "fragments",
 	    &stats->rx_fragments, "Fragmented frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_64",
 	    &stats->rx_pkts_64, "64 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_65_127",
 	    &stats->rx_pkts_65_127, "65 to 127 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_128_255",
 	    &stats->rx_pkts_128_255, "128 to 255 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_256_511",
 	    &stats->rx_pkts_256_511, "256 to 511 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_512_1023",
 	    &stats->rx_pkts_512_1023, "512 to 1023 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_1024_1518",
 	    &stats->rx_pkts_1024_1518, "1024 to 1518 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_1519_max",
 	    &stats->rx_pkts_1519_max, "1519 to max frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "trunc_errs",
 	    &stats->rx_pkts_truncated, "Truncated frames due to MTU size");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "fifo_oflows",
 	    &stats->rx_fifo_oflows, "FIFO overflows");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "rrs_errs",
 	    &stats->rx_rrs_errs, "Return status write-back errors");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "align_errs",
 	    &stats->rx_alignerrs, "Alignment errors");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "filtered",
 	    &stats->rx_pkts_filtered,
 	    "Frames dropped due to address filtering");
 
 	/* Tx statistics. */
 	tree = SYSCTL_ADD_NODE(ctx, parent, OID_AUTO, "tx", CTLFLAG_RD,
 	    NULL, "Tx MAC statistics");
 	child = SYSCTL_CHILDREN(tree);
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "good_frames",
 	    &stats->tx_frames, "Good frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "good_bcast_frames",
 	    &stats->tx_bcast_frames, "Good broadcast frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "good_mcast_frames",
 	    &stats->tx_mcast_frames, "Good multicast frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "pause_frames",
 	    &stats->tx_pause_frames, "Pause control frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "control_frames",
 	    &stats->tx_control_frames, "Control frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "excess_defers",
 	    &stats->tx_excess_defer, "Frames with excessive derferrals");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "defers",
 	    &stats->tx_excess_defer, "Frames with derferrals");
 	ALC_SYSCTL_STAT_ADD64(ctx, child, "good_octets",
 	    &stats->tx_bytes, "Good octets");
 	ALC_SYSCTL_STAT_ADD64(ctx, child, "good_bcast_octets",
 	    &stats->tx_bcast_bytes, "Good broadcast octets");
 	ALC_SYSCTL_STAT_ADD64(ctx, child, "good_mcast_octets",
 	    &stats->tx_mcast_bytes, "Good multicast octets");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_64",
 	    &stats->tx_pkts_64, "64 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_65_127",
 	    &stats->tx_pkts_65_127, "65 to 127 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_128_255",
 	    &stats->tx_pkts_128_255, "128 to 255 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_256_511",
 	    &stats->tx_pkts_256_511, "256 to 511 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_512_1023",
 	    &stats->tx_pkts_512_1023, "512 to 1023 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_1024_1518",
 	    &stats->tx_pkts_1024_1518, "1024 to 1518 bytes frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "frames_1519_max",
 	    &stats->tx_pkts_1519_max, "1519 to max frames");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "single_colls",
 	    &stats->tx_single_colls, "Single collisions");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "multi_colls",
 	    &stats->tx_multi_colls, "Multiple collisions");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "late_colls",
 	    &stats->tx_late_colls, "Late collisions");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "excess_colls",
 	    &stats->tx_excess_colls, "Excessive collisions");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "underruns",
 	    &stats->tx_underrun, "FIFO underruns");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "desc_underruns",
 	    &stats->tx_desc_underrun, "Descriptor write-back errors");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "len_errs",
 	    &stats->tx_lenerrs, "Frames with length mismatched");
 	ALC_SYSCTL_STAT_ADD32(ctx, child, "trunc_errs",
 	    &stats->tx_pkts_truncated, "Truncated frames due to MTU size");
 }
 
 #undef ALC_SYSCTL_STAT_ADD32
 #undef ALC_SYSCTL_STAT_ADD64
 
 struct alc_dmamap_arg {
 	bus_addr_t	alc_busaddr;
 };
 
 static void
 alc_dmamap_cb(void *arg, bus_dma_segment_t *segs, int nsegs, int error)
 {
 	struct alc_dmamap_arg *ctx;
 
 	if (error != 0)
 		return;
 
 	KASSERT(nsegs == 1, ("%s: %d segments returned!", __func__, nsegs));
 
 	ctx = (struct alc_dmamap_arg *)arg;
 	ctx->alc_busaddr = segs[0].ds_addr;
 }
 
 /*
  * Normal and high Tx descriptors shares single Tx high address.
  * Four Rx descriptor/return rings and CMB shares the same Rx
  * high address.
  */
 static int
 alc_check_boundary(struct alc_softc *sc)
 {
 	bus_addr_t cmb_end, rx_ring_end, rr_ring_end, tx_ring_end;
 
 	rx_ring_end = sc->alc_rdata.alc_rx_ring_paddr + ALC_RX_RING_SZ;
 	rr_ring_end = sc->alc_rdata.alc_rr_ring_paddr + ALC_RR_RING_SZ;
 	cmb_end = sc->alc_rdata.alc_cmb_paddr + ALC_CMB_SZ;
 	tx_ring_end = sc->alc_rdata.alc_tx_ring_paddr + ALC_TX_RING_SZ;
 
 	/* 4GB boundary crossing is not allowed. */
 	if ((ALC_ADDR_HI(rx_ring_end) !=
 	    ALC_ADDR_HI(sc->alc_rdata.alc_rx_ring_paddr)) ||
 	    (ALC_ADDR_HI(rr_ring_end) !=
 	    ALC_ADDR_HI(sc->alc_rdata.alc_rr_ring_paddr)) ||
 	    (ALC_ADDR_HI(cmb_end) !=
 	    ALC_ADDR_HI(sc->alc_rdata.alc_cmb_paddr)) ||
 	    (ALC_ADDR_HI(tx_ring_end) !=
 	    ALC_ADDR_HI(sc->alc_rdata.alc_tx_ring_paddr)))
 		return (EFBIG);
 	/*
 	 * Make sure Rx return descriptor/Rx descriptor/CMB use
 	 * the same high address.
 	 */
 	if ((ALC_ADDR_HI(rx_ring_end) != ALC_ADDR_HI(rr_ring_end)) ||
 	    (ALC_ADDR_HI(rx_ring_end) != ALC_ADDR_HI(cmb_end)))
 		return (EFBIG);
 
 	return (0);
 }
 
 static int
 alc_dma_alloc(struct alc_softc *sc)
 {
 	struct alc_txdesc *txd;
 	struct alc_rxdesc *rxd;
 	bus_addr_t lowaddr;
 	struct alc_dmamap_arg ctx;
 	int error, i;
 
 	lowaddr = BUS_SPACE_MAXADDR;
 again:
 	/* Create parent DMA tag. */
 	error = bus_dma_tag_create(
 	    bus_get_dma_tag(sc->alc_dev), /* parent */
 	    1, 0,			/* alignment, boundary */
 	    lowaddr,			/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    BUS_SPACE_MAXSIZE_32BIT,	/* maxsize */
 	    0,				/* nsegments */
 	    BUS_SPACE_MAXSIZE_32BIT,	/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_parent_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create parent DMA tag.\n");
 		goto fail;
 	}
 
 	/* Create DMA tag for Tx descriptor ring. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_parent_tag, /* parent */
 	    ALC_TX_RING_ALIGN, 0,	/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    ALC_TX_RING_SZ,		/* maxsize */
 	    1,				/* nsegments */
 	    ALC_TX_RING_SZ,		/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_tx_ring_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create Tx ring DMA tag.\n");
 		goto fail;
 	}
 
 	/* Create DMA tag for Rx free descriptor ring. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_parent_tag, /* parent */
 	    ALC_RX_RING_ALIGN, 0,	/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    ALC_RX_RING_SZ,		/* maxsize */
 	    1,				/* nsegments */
 	    ALC_RX_RING_SZ,		/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_rx_ring_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create Rx ring DMA tag.\n");
 		goto fail;
 	}
 	/* Create DMA tag for Rx return descriptor ring. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_parent_tag, /* parent */
 	    ALC_RR_RING_ALIGN, 0,	/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    ALC_RR_RING_SZ,		/* maxsize */
 	    1,				/* nsegments */
 	    ALC_RR_RING_SZ,		/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_rr_ring_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create Rx return ring DMA tag.\n");
 		goto fail;
 	}
 
 	/* Create DMA tag for coalescing message block. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_parent_tag, /* parent */
 	    ALC_CMB_ALIGN, 0,		/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    ALC_CMB_SZ,			/* maxsize */
 	    1,				/* nsegments */
 	    ALC_CMB_SZ,			/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_cmb_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create CMB DMA tag.\n");
 		goto fail;
 	}
 	/* Create DMA tag for status message block. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_parent_tag, /* parent */
 	    ALC_SMB_ALIGN, 0,		/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    ALC_SMB_SZ,			/* maxsize */
 	    1,				/* nsegments */
 	    ALC_SMB_SZ,			/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_smb_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create SMB DMA tag.\n");
 		goto fail;
 	}
 
 	/* Allocate DMA'able memory and load the DMA map for Tx ring. */
 	error = bus_dmamem_alloc(sc->alc_cdata.alc_tx_ring_tag,
 	    (void **)&sc->alc_rdata.alc_tx_ring,
 	    BUS_DMA_WAITOK | BUS_DMA_ZERO | BUS_DMA_COHERENT,
 	    &sc->alc_cdata.alc_tx_ring_map);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not allocate DMA'able memory for Tx ring.\n");
 		goto fail;
 	}
 	ctx.alc_busaddr = 0;
 	error = bus_dmamap_load(sc->alc_cdata.alc_tx_ring_tag,
 	    sc->alc_cdata.alc_tx_ring_map, sc->alc_rdata.alc_tx_ring,
 	    ALC_TX_RING_SZ, alc_dmamap_cb, &ctx, 0);
 	if (error != 0 || ctx.alc_busaddr == 0) {
 		device_printf(sc->alc_dev,
 		    "could not load DMA'able memory for Tx ring.\n");
 		goto fail;
 	}
 	sc->alc_rdata.alc_tx_ring_paddr = ctx.alc_busaddr;
 
 	/* Allocate DMA'able memory and load the DMA map for Rx ring. */
 	error = bus_dmamem_alloc(sc->alc_cdata.alc_rx_ring_tag,
 	    (void **)&sc->alc_rdata.alc_rx_ring,
 	    BUS_DMA_WAITOK | BUS_DMA_ZERO | BUS_DMA_COHERENT,
 	    &sc->alc_cdata.alc_rx_ring_map);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not allocate DMA'able memory for Rx ring.\n");
 		goto fail;
 	}
 	ctx.alc_busaddr = 0;
 	error = bus_dmamap_load(sc->alc_cdata.alc_rx_ring_tag,
 	    sc->alc_cdata.alc_rx_ring_map, sc->alc_rdata.alc_rx_ring,
 	    ALC_RX_RING_SZ, alc_dmamap_cb, &ctx, 0);
 	if (error != 0 || ctx.alc_busaddr == 0) {
 		device_printf(sc->alc_dev,
 		    "could not load DMA'able memory for Rx ring.\n");
 		goto fail;
 	}
 	sc->alc_rdata.alc_rx_ring_paddr = ctx.alc_busaddr;
 
 	/* Allocate DMA'able memory and load the DMA map for Rx return ring. */
 	error = bus_dmamem_alloc(sc->alc_cdata.alc_rr_ring_tag,
 	    (void **)&sc->alc_rdata.alc_rr_ring,
 	    BUS_DMA_WAITOK | BUS_DMA_ZERO | BUS_DMA_COHERENT,
 	    &sc->alc_cdata.alc_rr_ring_map);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not allocate DMA'able memory for Rx return ring.\n");
 		goto fail;
 	}
 	ctx.alc_busaddr = 0;
 	error = bus_dmamap_load(sc->alc_cdata.alc_rr_ring_tag,
 	    sc->alc_cdata.alc_rr_ring_map, sc->alc_rdata.alc_rr_ring,
 	    ALC_RR_RING_SZ, alc_dmamap_cb, &ctx, 0);
 	if (error != 0 || ctx.alc_busaddr == 0) {
 		device_printf(sc->alc_dev,
 		    "could not load DMA'able memory for Tx ring.\n");
 		goto fail;
 	}
 	sc->alc_rdata.alc_rr_ring_paddr = ctx.alc_busaddr;
 
 	/* Allocate DMA'able memory and load the DMA map for CMB. */
 	error = bus_dmamem_alloc(sc->alc_cdata.alc_cmb_tag,
 	    (void **)&sc->alc_rdata.alc_cmb,
 	    BUS_DMA_WAITOK | BUS_DMA_ZERO | BUS_DMA_COHERENT,
 	    &sc->alc_cdata.alc_cmb_map);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not allocate DMA'able memory for CMB.\n");
 		goto fail;
 	}
 	ctx.alc_busaddr = 0;
 	error = bus_dmamap_load(sc->alc_cdata.alc_cmb_tag,
 	    sc->alc_cdata.alc_cmb_map, sc->alc_rdata.alc_cmb,
 	    ALC_CMB_SZ, alc_dmamap_cb, &ctx, 0);
 	if (error != 0 || ctx.alc_busaddr == 0) {
 		device_printf(sc->alc_dev,
 		    "could not load DMA'able memory for CMB.\n");
 		goto fail;
 	}
 	sc->alc_rdata.alc_cmb_paddr = ctx.alc_busaddr;
 
 	/* Allocate DMA'able memory and load the DMA map for SMB. */
 	error = bus_dmamem_alloc(sc->alc_cdata.alc_smb_tag,
 	    (void **)&sc->alc_rdata.alc_smb,
 	    BUS_DMA_WAITOK | BUS_DMA_ZERO | BUS_DMA_COHERENT,
 	    &sc->alc_cdata.alc_smb_map);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not allocate DMA'able memory for SMB.\n");
 		goto fail;
 	}
 	ctx.alc_busaddr = 0;
 	error = bus_dmamap_load(sc->alc_cdata.alc_smb_tag,
 	    sc->alc_cdata.alc_smb_map, sc->alc_rdata.alc_smb,
 	    ALC_SMB_SZ, alc_dmamap_cb, &ctx, 0);
 	if (error != 0 || ctx.alc_busaddr == 0) {
 		device_printf(sc->alc_dev,
 		    "could not load DMA'able memory for CMB.\n");
 		goto fail;
 	}
 	sc->alc_rdata.alc_smb_paddr = ctx.alc_busaddr;
 
 	/* Make sure we've not crossed 4GB boundary. */
 	if (lowaddr != BUS_SPACE_MAXADDR_32BIT &&
 	    (error = alc_check_boundary(sc)) != 0) {
 		device_printf(sc->alc_dev, "4GB boundary crossed, "
 		    "switching to 32bit DMA addressing mode.\n");
 		alc_dma_free(sc);
 		/*
 		 * Limit max allowable DMA address space to 32bit
 		 * and try again.
 		 */
 		lowaddr = BUS_SPACE_MAXADDR_32BIT;
 		goto again;
 	}
 
 	/*
 	 * Create Tx buffer parent tag.
 	 * AR81[3567]x allows 64bit DMA addressing of Tx/Rx buffers
 	 * so it needs separate parent DMA tag as parent DMA address
 	 * space could be restricted to be within 32bit address space
 	 * by 4GB boundary crossing.
 	 */
 	error = bus_dma_tag_create(
 	    bus_get_dma_tag(sc->alc_dev), /* parent */
 	    1, 0,			/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    BUS_SPACE_MAXSIZE_32BIT,	/* maxsize */
 	    0,				/* nsegments */
 	    BUS_SPACE_MAXSIZE_32BIT,	/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_buffer_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create parent buffer DMA tag.\n");
 		goto fail;
 	}
 
 	/* Create DMA tag for Tx buffers. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_buffer_tag, /* parent */
 	    1, 0,			/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    ALC_TSO_MAXSIZE,		/* maxsize */
 	    ALC_MAXTXSEGS,		/* nsegments */
 	    ALC_TSO_MAXSEGSIZE,		/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_tx_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev, "could not create Tx DMA tag.\n");
 		goto fail;
 	}
 
 	/* Create DMA tag for Rx buffers. */
 	error = bus_dma_tag_create(
 	    sc->alc_cdata.alc_buffer_tag, /* parent */
 	    ALC_RX_BUF_ALIGN, 0,	/* alignment, boundary */
 	    BUS_SPACE_MAXADDR,		/* lowaddr */
 	    BUS_SPACE_MAXADDR,		/* highaddr */
 	    NULL, NULL,			/* filter, filterarg */
 	    MCLBYTES,			/* maxsize */
 	    1,				/* nsegments */
 	    MCLBYTES,			/* maxsegsize */
 	    0,				/* flags */
 	    NULL, NULL,			/* lockfunc, lockarg */
 	    &sc->alc_cdata.alc_rx_tag);
 	if (error != 0) {
 		device_printf(sc->alc_dev, "could not create Rx DMA tag.\n");
 		goto fail;
 	}
 	/* Create DMA maps for Tx buffers. */
 	for (i = 0; i < ALC_TX_RING_CNT; i++) {
 		txd = &sc->alc_cdata.alc_txdesc[i];
 		txd->tx_m = NULL;
 		txd->tx_dmamap = NULL;
 		error = bus_dmamap_create(sc->alc_cdata.alc_tx_tag, 0,
 		    &txd->tx_dmamap);
 		if (error != 0) {
 			device_printf(sc->alc_dev,
 			    "could not create Tx dmamap.\n");
 			goto fail;
 		}
 	}
 	/* Create DMA maps for Rx buffers. */
 	if ((error = bus_dmamap_create(sc->alc_cdata.alc_rx_tag, 0,
 	    &sc->alc_cdata.alc_rx_sparemap)) != 0) {
 		device_printf(sc->alc_dev,
 		    "could not create spare Rx dmamap.\n");
 		goto fail;
 	}
 	for (i = 0; i < ALC_RX_RING_CNT; i++) {
 		rxd = &sc->alc_cdata.alc_rxdesc[i];
 		rxd->rx_m = NULL;
 		rxd->rx_dmamap = NULL;
 		error = bus_dmamap_create(sc->alc_cdata.alc_rx_tag, 0,
 		    &rxd->rx_dmamap);
 		if (error != 0) {
 			device_printf(sc->alc_dev,
 			    "could not create Rx dmamap.\n");
 			goto fail;
 		}
 	}
 
 fail:
 	return (error);
 }
 
 static void
 alc_dma_free(struct alc_softc *sc)
 {
 	struct alc_txdesc *txd;
 	struct alc_rxdesc *rxd;
 	int i;
 
 	/* Tx buffers. */
 	if (sc->alc_cdata.alc_tx_tag != NULL) {
 		for (i = 0; i < ALC_TX_RING_CNT; i++) {
 			txd = &sc->alc_cdata.alc_txdesc[i];
 			if (txd->tx_dmamap != NULL) {
 				bus_dmamap_destroy(sc->alc_cdata.alc_tx_tag,
 				    txd->tx_dmamap);
 				txd->tx_dmamap = NULL;
 			}
 		}
 		bus_dma_tag_destroy(sc->alc_cdata.alc_tx_tag);
 		sc->alc_cdata.alc_tx_tag = NULL;
 	}
 	/* Rx buffers */
 	if (sc->alc_cdata.alc_rx_tag != NULL) {
 		for (i = 0; i < ALC_RX_RING_CNT; i++) {
 			rxd = &sc->alc_cdata.alc_rxdesc[i];
 			if (rxd->rx_dmamap != NULL) {
 				bus_dmamap_destroy(sc->alc_cdata.alc_rx_tag,
 				    rxd->rx_dmamap);
 				rxd->rx_dmamap = NULL;
 			}
 		}
 		if (sc->alc_cdata.alc_rx_sparemap != NULL) {
 			bus_dmamap_destroy(sc->alc_cdata.alc_rx_tag,
 			    sc->alc_cdata.alc_rx_sparemap);
 			sc->alc_cdata.alc_rx_sparemap = NULL;
 		}
 		bus_dma_tag_destroy(sc->alc_cdata.alc_rx_tag);
 		sc->alc_cdata.alc_rx_tag = NULL;
 	}
 	/* Tx descriptor ring. */
 	if (sc->alc_cdata.alc_tx_ring_tag != NULL) {
 		if (sc->alc_rdata.alc_tx_ring_paddr != 0)
 			bus_dmamap_unload(sc->alc_cdata.alc_tx_ring_tag,
 			    sc->alc_cdata.alc_tx_ring_map);
 		if (sc->alc_rdata.alc_tx_ring != NULL)
 			bus_dmamem_free(sc->alc_cdata.alc_tx_ring_tag,
 			    sc->alc_rdata.alc_tx_ring,
 			    sc->alc_cdata.alc_tx_ring_map);
 		sc->alc_rdata.alc_tx_ring_paddr = 0;
 		sc->alc_rdata.alc_tx_ring = NULL;
 		bus_dma_tag_destroy(sc->alc_cdata.alc_tx_ring_tag);
 		sc->alc_cdata.alc_tx_ring_tag = NULL;
 	}
 	/* Rx ring. */
 	if (sc->alc_cdata.alc_rx_ring_tag != NULL) {
 		if (sc->alc_rdata.alc_rx_ring_paddr != 0)
 			bus_dmamap_unload(sc->alc_cdata.alc_rx_ring_tag,
 			    sc->alc_cdata.alc_rx_ring_map);
 		if (sc->alc_rdata.alc_rx_ring != NULL)
 			bus_dmamem_free(sc->alc_cdata.alc_rx_ring_tag,
 			    sc->alc_rdata.alc_rx_ring,
 			    sc->alc_cdata.alc_rx_ring_map);
 		sc->alc_rdata.alc_rx_ring_paddr = 0;
 		sc->alc_rdata.alc_rx_ring = NULL;
 		bus_dma_tag_destroy(sc->alc_cdata.alc_rx_ring_tag);
 		sc->alc_cdata.alc_rx_ring_tag = NULL;
 	}
 	/* Rx return ring. */
 	if (sc->alc_cdata.alc_rr_ring_tag != NULL) {
 		if (sc->alc_rdata.alc_rr_ring_paddr != 0)
 			bus_dmamap_unload(sc->alc_cdata.alc_rr_ring_tag,
 			    sc->alc_cdata.alc_rr_ring_map);
 		if (sc->alc_rdata.alc_rr_ring != NULL)
 			bus_dmamem_free(sc->alc_cdata.alc_rr_ring_tag,
 			    sc->alc_rdata.alc_rr_ring,
 			    sc->alc_cdata.alc_rr_ring_map);
 		sc->alc_rdata.alc_rr_ring_paddr = 0;
 		sc->alc_rdata.alc_rr_ring = NULL;
 		bus_dma_tag_destroy(sc->alc_cdata.alc_rr_ring_tag);
 		sc->alc_cdata.alc_rr_ring_tag = NULL;
 	}
 	/* CMB block */
 	if (sc->alc_cdata.alc_cmb_tag != NULL) {
 		if (sc->alc_rdata.alc_cmb_paddr != 0)
 			bus_dmamap_unload(sc->alc_cdata.alc_cmb_tag,
 			    sc->alc_cdata.alc_cmb_map);
 		if (sc->alc_rdata.alc_cmb != NULL)
 			bus_dmamem_free(sc->alc_cdata.alc_cmb_tag,
 			    sc->alc_rdata.alc_cmb,
 			    sc->alc_cdata.alc_cmb_map);		
 		sc->alc_rdata.alc_cmb_paddr = 0;
 		sc->alc_rdata.alc_cmb = NULL;
 		bus_dma_tag_destroy(sc->alc_cdata.alc_cmb_tag);
 		sc->alc_cdata.alc_cmb_tag = NULL;
 	}
 	/* SMB block */
 	if (sc->alc_cdata.alc_smb_tag != NULL) {
 		if (sc->alc_rdata.alc_smb_paddr != 0)
 			bus_dmamap_unload(sc->alc_cdata.alc_smb_tag,
 			    sc->alc_cdata.alc_smb_map);
 		if (sc->alc_rdata.alc_smb != NULL)
 			bus_dmamem_free(sc->alc_cdata.alc_smb_tag,
 			    sc->alc_rdata.alc_smb,
 			    sc->alc_cdata.alc_smb_map);
 		sc->alc_rdata.alc_smb_paddr = 0;
 		sc->alc_rdata.alc_smb = NULL;
 		bus_dma_tag_destroy(sc->alc_cdata.alc_smb_tag);
 		sc->alc_cdata.alc_smb_tag = NULL;
 	}
 	if (sc->alc_cdata.alc_buffer_tag != NULL) {
 		bus_dma_tag_destroy(sc->alc_cdata.alc_buffer_tag);
 		sc->alc_cdata.alc_buffer_tag = NULL;
 	}
 	if (sc->alc_cdata.alc_parent_tag != NULL) {
 		bus_dma_tag_destroy(sc->alc_cdata.alc_parent_tag);
 		sc->alc_cdata.alc_parent_tag = NULL;
 	}
 }
 
 static int
 alc_shutdown(device_t dev)
 {
 
 	return (alc_suspend(dev));
 }
 
 /*
  * Note, this driver resets the link speed to 10/100Mbps by
  * restarting auto-negotiation in suspend/shutdown phase but we
  * don't know whether that auto-negotiation would succeed or not
  * as driver has no control after powering off/suspend operation.
  * If the renegotiation fail WOL may not work. Running at 1Gbps
  * will draw more power than 375mA at 3.3V which is specified in
  * PCI specification and that would result in complete
  * shutdowning power to ethernet controller.
  *
  * TODO
  * Save current negotiated media speed/duplex/flow-control to
  * softc and restore the same link again after resuming. PHY
  * handling such as power down/resetting to 100Mbps may be better
  * handled in suspend method in phy driver.
  */
 static void
 alc_setlinkspeed(struct alc_softc *sc)
 {
 	struct mii_data *mii;
 	int aneg, i;
 
 	mii = device_get_softc(sc->alc_miibus);
 	mii_pollstat(mii);
 	aneg = 0;
 	if ((mii->mii_media_status & (IFM_ACTIVE | IFM_AVALID)) ==
 	    (IFM_ACTIVE | IFM_AVALID)) {
 		switch IFM_SUBTYPE(mii->mii_media_active) {
 		case IFM_10_T:
 		case IFM_100_TX:
 			return;
 		case IFM_1000_T:
 			aneg++;
 			break;
 		default:
 			break;
 		}
 	}
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr, MII_100T2CR, 0);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    MII_ANAR, ANAR_TX_FD | ANAR_TX | ANAR_10_FD | ANAR_10 | ANAR_CSMA);
 	alc_miibus_writereg(sc->alc_dev, sc->alc_phyaddr,
 	    MII_BMCR, BMCR_RESET | BMCR_AUTOEN | BMCR_STARTNEG);
 	DELAY(1000);
 	if (aneg != 0) {
 		/*
 		 * Poll link state until alc(4) get a 10/100Mbps link.
 		 */
 		for (i = 0; i < MII_ANEGTICKS_GIGE; i++) {
 			mii_pollstat(mii);
 			if ((mii->mii_media_status & (IFM_ACTIVE | IFM_AVALID))
 			    == (IFM_ACTIVE | IFM_AVALID)) {
 				switch (IFM_SUBTYPE(
 				    mii->mii_media_active)) {
 				case IFM_10_T:
 				case IFM_100_TX:
 					alc_mac_config(sc);
 					return;
 				default:
 					break;
 				}
 			}
 			ALC_UNLOCK(sc);
 			pause("alclnk", hz);
 			ALC_LOCK(sc);
 		}
 		if (i == MII_ANEGTICKS_GIGE)
 			device_printf(sc->alc_dev,
 			    "establishing a link failed, WOL may not work!");
 	}
 	/*
 	 * No link, force MAC to have 100Mbps, full-duplex link.
 	 * This is the last resort and may/may not work.
 	 */
 	mii->mii_media_status = IFM_AVALID | IFM_ACTIVE;
 	mii->mii_media_active = IFM_ETHER | IFM_100_TX | IFM_FDX;
 	alc_mac_config(sc);
 }
 
 static void
 alc_setwol(struct alc_softc *sc)
 {
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		alc_setwol_816x(sc);
 	else
 		alc_setwol_813x(sc);
 }
 
 static void
 alc_setwol_813x(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	uint32_t reg, pmcs;
 	uint16_t pmstat;
 
 	ALC_LOCK_ASSERT(sc);
 
 	alc_disable_l0s_l1(sc);
 	ifp = sc->alc_ifp;
 	if ((sc->alc_flags & ALC_FLAG_PM) == 0) {
 		/* Disable WOL. */
 		CSR_WRITE_4(sc, ALC_WOL_CFG, 0);
 		reg = CSR_READ_4(sc, ALC_PCIE_PHYMISC);
 		reg |= PCIE_PHYMISC_FORCE_RCV_DET;
 		CSR_WRITE_4(sc, ALC_PCIE_PHYMISC, reg);
 		/* Force PHY power down. */
 		alc_phy_down(sc);
 		CSR_WRITE_4(sc, ALC_MASTER_CFG,
 		    CSR_READ_4(sc, ALC_MASTER_CFG) | MASTER_CLK_SEL_DIS);
 		return;
 	}
 
 	if ((ifp->if_capenable & IFCAP_WOL) != 0) {
 		if ((sc->alc_flags & ALC_FLAG_FASTETHER) == 0)
 			alc_setlinkspeed(sc);
 		CSR_WRITE_4(sc, ALC_MASTER_CFG,
 		    CSR_READ_4(sc, ALC_MASTER_CFG) & ~MASTER_CLK_SEL_DIS);
 	}
 
 	pmcs = 0;
 	if ((ifp->if_capenable & IFCAP_WOL_MAGIC) != 0)
 		pmcs |= WOL_CFG_MAGIC | WOL_CFG_MAGIC_ENB;
 	CSR_WRITE_4(sc, ALC_WOL_CFG, pmcs);
 	reg = CSR_READ_4(sc, ALC_MAC_CFG);
 	reg &= ~(MAC_CFG_DBG | MAC_CFG_PROMISC | MAC_CFG_ALLMULTI |
 	    MAC_CFG_BCAST);
 	if ((ifp->if_capenable & IFCAP_WOL_MCAST) != 0)
 		reg |= MAC_CFG_ALLMULTI | MAC_CFG_BCAST;
 	if ((ifp->if_capenable & IFCAP_WOL) != 0)
 		reg |= MAC_CFG_RX_ENB;
 	CSR_WRITE_4(sc, ALC_MAC_CFG, reg);
 
 	reg = CSR_READ_4(sc, ALC_PCIE_PHYMISC);
 	reg |= PCIE_PHYMISC_FORCE_RCV_DET;
 	CSR_WRITE_4(sc, ALC_PCIE_PHYMISC, reg);
 	if ((ifp->if_capenable & IFCAP_WOL) == 0) {
 		/* WOL disabled, PHY power down. */
 		alc_phy_down(sc);
 		CSR_WRITE_4(sc, ALC_MASTER_CFG,
 		    CSR_READ_4(sc, ALC_MASTER_CFG) | MASTER_CLK_SEL_DIS);
 	}
 	/* Request PME. */
 	pmstat = pci_read_config(sc->alc_dev,
 	    sc->alc_pmcap + PCIR_POWER_STATUS, 2);
 	pmstat &= ~(PCIM_PSTAT_PME | PCIM_PSTAT_PMEENABLE);
 	if ((ifp->if_capenable & IFCAP_WOL) != 0)
 		pmstat |= PCIM_PSTAT_PME | PCIM_PSTAT_PMEENABLE;
 	pci_write_config(sc->alc_dev,
 	    sc->alc_pmcap + PCIR_POWER_STATUS, pmstat, 2);
 }
 
 static void
 alc_setwol_816x(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	uint32_t gphy, mac, master, pmcs, reg;
 	uint16_t pmstat;
 
 	ALC_LOCK_ASSERT(sc);
 
 	ifp = sc->alc_ifp;
 	master = CSR_READ_4(sc, ALC_MASTER_CFG);
 	master &= ~MASTER_CLK_SEL_DIS;
 	gphy = CSR_READ_4(sc, ALC_GPHY_CFG);
 	gphy &= ~(GPHY_CFG_EXT_RESET | GPHY_CFG_LED_MODE | GPHY_CFG_100AB_ENB |
 	    GPHY_CFG_PHY_PLL_ON);
 	gphy |= GPHY_CFG_HIB_EN | GPHY_CFG_HIB_PULSE | GPHY_CFG_SEL_ANA_RESET;
 	if ((sc->alc_flags & ALC_FLAG_PM) == 0) {
 		CSR_WRITE_4(sc, ALC_WOL_CFG, 0);
 		gphy |= GPHY_CFG_PHY_IDDQ | GPHY_CFG_PWDOWN_HW;
 		mac = CSR_READ_4(sc, ALC_MAC_CFG);
 	} else {
 		if ((ifp->if_capenable & IFCAP_WOL) != 0) {
 			gphy |= GPHY_CFG_EXT_RESET;
 			if ((sc->alc_flags & ALC_FLAG_FASTETHER) == 0)
 				alc_setlinkspeed(sc);
 		}
 		pmcs = 0;
 		if ((ifp->if_capenable & IFCAP_WOL_MAGIC) != 0)
 			pmcs |= WOL_CFG_MAGIC | WOL_CFG_MAGIC_ENB;
 		CSR_WRITE_4(sc, ALC_WOL_CFG, pmcs);
 		mac = CSR_READ_4(sc, ALC_MAC_CFG);
 		mac &= ~(MAC_CFG_DBG | MAC_CFG_PROMISC | MAC_CFG_ALLMULTI |
 		    MAC_CFG_BCAST);
 		if ((ifp->if_capenable & IFCAP_WOL_MCAST) != 0)
 			mac |= MAC_CFG_ALLMULTI | MAC_CFG_BCAST;
 		if ((ifp->if_capenable & IFCAP_WOL) != 0)
 			mac |= MAC_CFG_RX_ENB;
 		alc_miiext_writereg(sc, MII_EXT_ANEG, MII_EXT_ANEG_S3DIG10,
 		    ANEG_S3DIG10_SL);
 	}
 
 	/* Enable OSC. */
 	reg = CSR_READ_4(sc, ALC_MISC);
 	reg &= ~MISC_INTNLOSC_OPEN;
 	CSR_WRITE_4(sc, ALC_MISC, reg);
 	reg |= MISC_INTNLOSC_OPEN;
 	CSR_WRITE_4(sc, ALC_MISC, reg);
 	CSR_WRITE_4(sc, ALC_MASTER_CFG, master);
 	CSR_WRITE_4(sc, ALC_MAC_CFG, mac);
 	CSR_WRITE_4(sc, ALC_GPHY_CFG, gphy);
 	reg = CSR_READ_4(sc, ALC_PDLL_TRNS1);
 	reg |= PDLL_TRNS1_D3PLLOFF_ENB;
 	CSR_WRITE_4(sc, ALC_PDLL_TRNS1, reg);
 
 	if ((sc->alc_flags & ALC_FLAG_PM) != 0) {
 		/* Request PME. */
 		pmstat = pci_read_config(sc->alc_dev,
 		    sc->alc_pmcap + PCIR_POWER_STATUS, 2);
 		pmstat &= ~(PCIM_PSTAT_PME | PCIM_PSTAT_PMEENABLE);
 		if ((ifp->if_capenable & IFCAP_WOL) != 0)
 			pmstat |= PCIM_PSTAT_PME | PCIM_PSTAT_PMEENABLE;
 		pci_write_config(sc->alc_dev,
 		    sc->alc_pmcap + PCIR_POWER_STATUS, pmstat, 2);
 	}
 }
 
 static int
 alc_suspend(device_t dev)
 {
 	struct alc_softc *sc;
 
 	sc = device_get_softc(dev);
 
 	ALC_LOCK(sc);
 	alc_stop(sc);
 	alc_setwol(sc);
 	ALC_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
 alc_resume(device_t dev)
 {
 	struct alc_softc *sc;
 	struct ifnet *ifp;
 	uint16_t pmstat;
 
 	sc = device_get_softc(dev);
 
 	ALC_LOCK(sc);
 	if ((sc->alc_flags & ALC_FLAG_PM) != 0) {
 		/* Disable PME and clear PME status. */
 		pmstat = pci_read_config(sc->alc_dev,
 		    sc->alc_pmcap + PCIR_POWER_STATUS, 2);
 		if ((pmstat & PCIM_PSTAT_PMEENABLE) != 0) {
 			pmstat &= ~PCIM_PSTAT_PMEENABLE;
 			pci_write_config(sc->alc_dev,
 			    sc->alc_pmcap + PCIR_POWER_STATUS, pmstat, 2);
 		}
 	}
 	/* Reset PHY. */
 	alc_phy_reset(sc);
 	ifp = sc->alc_ifp;
 	if ((ifp->if_flags & IFF_UP) != 0) {
 		ifp->if_drv_flags &= ~IFF_DRV_RUNNING;
 		alc_init_locked(sc);
 	}
 	ALC_UNLOCK(sc);
 
 	return (0);
 }
 
 static int
 alc_encap(struct alc_softc *sc, struct mbuf **m_head)
 {
 	struct alc_txdesc *txd, *txd_last;
 	struct tx_desc *desc;
 	struct mbuf *m;
 	struct ip *ip;
 	struct tcphdr *tcp;
 	bus_dma_segment_t txsegs[ALC_MAXTXSEGS];
 	bus_dmamap_t map;
 	uint32_t cflags, hdrlen, ip_off, poff, vtag;
 	int error, idx, nsegs, prod;
 
 	ALC_LOCK_ASSERT(sc);
 
 	M_ASSERTPKTHDR((*m_head));
 
 	m = *m_head;
 	ip = NULL;
 	tcp = NULL;
 	ip_off = poff = 0;
 	if ((m->m_pkthdr.csum_flags & (ALC_CSUM_FEATURES | CSUM_TSO)) != 0) {
 		/*
 		 * AR81[3567]x requires offset of TCP/UDP header in its
 		 * Tx descriptor to perform Tx checksum offloading. TSO
 		 * also requires TCP header offset and modification of
 		 * IP/TCP header. This kind of operation takes many CPU
 		 * cycles on FreeBSD so fast host CPU is required to get
 		 * smooth TSO performance.
 		 */
 		struct ether_header *eh;
 
 		if (M_WRITABLE(m) == 0) {
 			/* Get a writable copy. */
 			m = m_dup(*m_head, M_NOWAIT);
 			/* Release original mbufs. */
 			m_freem(*m_head);
 			if (m == NULL) {
 				*m_head = NULL;
 				return (ENOBUFS);
 			}
 			*m_head = m;
 		}
 
 		ip_off = sizeof(struct ether_header);
 		m = m_pullup(m, ip_off);
 		if (m == NULL) {
 			*m_head = NULL;
 			return (ENOBUFS);
 		}
 		eh = mtod(m, struct ether_header *);
 		/*
 		 * Check if hardware VLAN insertion is off.
 		 * Additional check for LLC/SNAP frame?
 		 */
 		if (eh->ether_type == htons(ETHERTYPE_VLAN)) {
 			ip_off = sizeof(struct ether_vlan_header);
 			m = m_pullup(m, ip_off);
 			if (m == NULL) {
 				*m_head = NULL;
 				return (ENOBUFS);
 			}
 		}
 		m = m_pullup(m, ip_off + sizeof(struct ip));
 		if (m == NULL) {
 			*m_head = NULL;
 			return (ENOBUFS);
 		}
 		ip = (struct ip *)(mtod(m, char *) + ip_off);
 		poff = ip_off + (ip->ip_hl << 2);
 		if ((m->m_pkthdr.csum_flags & CSUM_TSO) != 0) {
 			m = m_pullup(m, poff + sizeof(struct tcphdr));
 			if (m == NULL) {
 				*m_head = NULL;
 				return (ENOBUFS);
 			}
 			tcp = (struct tcphdr *)(mtod(m, char *) + poff);
 			m = m_pullup(m, poff + (tcp->th_off << 2));
 			if (m == NULL) {
 				*m_head = NULL;
 				return (ENOBUFS);
 			}
 			/*
 			 * Due to strict adherence of Microsoft NDIS
 			 * Large Send specification, hardware expects
 			 * a pseudo TCP checksum inserted by upper
 			 * stack. Unfortunately the pseudo TCP
 			 * checksum that NDIS refers to does not include
 			 * TCP payload length so driver should recompute
 			 * the pseudo checksum here. Hopefully this
 			 * wouldn't be much burden on modern CPUs.
 			 *
 			 * Reset IP checksum and recompute TCP pseudo
 			 * checksum as NDIS specification said.
 			 */
 			ip = (struct ip *)(mtod(m, char *) + ip_off);
 			tcp = (struct tcphdr *)(mtod(m, char *) + poff);
 			ip->ip_sum = 0;
 			tcp->th_sum = in_pseudo(ip->ip_src.s_addr,
 			    ip->ip_dst.s_addr, htons(IPPROTO_TCP));
 		}
 		*m_head = m;
 	}
 
 	prod = sc->alc_cdata.alc_tx_prod;
 	txd = &sc->alc_cdata.alc_txdesc[prod];
 	txd_last = txd;
 	map = txd->tx_dmamap;
 
 	error = bus_dmamap_load_mbuf_sg(sc->alc_cdata.alc_tx_tag, map,
 	    *m_head, txsegs, &nsegs, 0);
 	if (error == EFBIG) {
 		m = m_collapse(*m_head, M_NOWAIT, ALC_MAXTXSEGS);
 		if (m == NULL) {
 			m_freem(*m_head);
 			*m_head = NULL;
 			return (ENOMEM);
 		}
 		*m_head = m;
 		error = bus_dmamap_load_mbuf_sg(sc->alc_cdata.alc_tx_tag, map,
 		    *m_head, txsegs, &nsegs, 0);
 		if (error != 0) {
 			m_freem(*m_head);
 			*m_head = NULL;
 			return (error);
 		}
 	} else if (error != 0)
 		return (error);
 	if (nsegs == 0) {
 		m_freem(*m_head);
 		*m_head = NULL;
 		return (EIO);
 	}
 
 	/* Check descriptor overrun. */
 	if (sc->alc_cdata.alc_tx_cnt + nsegs >= ALC_TX_RING_CNT - 3) {
 		bus_dmamap_unload(sc->alc_cdata.alc_tx_tag, map);
 		return (ENOBUFS);
 	}
 	bus_dmamap_sync(sc->alc_cdata.alc_tx_tag, map, BUS_DMASYNC_PREWRITE);
 
 	m = *m_head;
 	cflags = TD_ETHERNET;
 	vtag = 0;
 	desc = NULL;
 	idx = 0;
 	/* Configure VLAN hardware tag insertion. */
 	if ((m->m_flags & M_VLANTAG) != 0) {
 		vtag = htons(m->m_pkthdr.ether_vtag);
 		vtag = (vtag << TD_VLAN_SHIFT) & TD_VLAN_MASK;
 		cflags |= TD_INS_VLAN_TAG;
 	}
 	if ((m->m_pkthdr.csum_flags & CSUM_TSO) != 0) {
 		/* Request TSO and set MSS. */
 		cflags |= TD_TSO | TD_TSO_DESCV1;
 		cflags |= ((uint32_t)m->m_pkthdr.tso_segsz << TD_MSS_SHIFT) &
 		    TD_MSS_MASK;
 		/* Set TCP header offset. */
 		cflags |= (poff << TD_TCPHDR_OFFSET_SHIFT) &
 		    TD_TCPHDR_OFFSET_MASK;
 		/*
 		 * AR81[3567]x requires the first buffer should
 		 * only hold IP/TCP header data. Payload should
 		 * be handled in other descriptors.
 		 */
 		hdrlen = poff + (tcp->th_off << 2);
 		desc = &sc->alc_rdata.alc_tx_ring[prod];
 		desc->len = htole32(TX_BYTES(hdrlen | vtag));
 		desc->flags = htole32(cflags);
 		desc->addr = htole64(txsegs[0].ds_addr);
 		sc->alc_cdata.alc_tx_cnt++;
 		ALC_DESC_INC(prod, ALC_TX_RING_CNT);
 		if (m->m_len - hdrlen > 0) {
 			/* Handle remaining payload of the first fragment. */
 			desc = &sc->alc_rdata.alc_tx_ring[prod];
 			desc->len = htole32(TX_BYTES((m->m_len - hdrlen) |
 			    vtag));
 			desc->flags = htole32(cflags);
 			desc->addr = htole64(txsegs[0].ds_addr + hdrlen);
 			sc->alc_cdata.alc_tx_cnt++;
 			ALC_DESC_INC(prod, ALC_TX_RING_CNT);
 		}
 		/* Handle remaining fragments. */
 		idx = 1;
 	} else if ((m->m_pkthdr.csum_flags & ALC_CSUM_FEATURES) != 0) {
 		/* Configure Tx checksum offload. */
 #ifdef ALC_USE_CUSTOM_CSUM
 		cflags |= TD_CUSTOM_CSUM;
 		/* Set checksum start offset. */
 		cflags |= ((poff >> 1) << TD_PLOAD_OFFSET_SHIFT) &
 		    TD_PLOAD_OFFSET_MASK;
 		/* Set checksum insertion position of TCP/UDP. */
 		cflags |= (((poff + m->m_pkthdr.csum_data) >> 1) <<
 		    TD_CUSTOM_CSUM_OFFSET_SHIFT) & TD_CUSTOM_CSUM_OFFSET_MASK;
 #else
 		if ((m->m_pkthdr.csum_flags & CSUM_IP) != 0)
 			cflags |= TD_IPCSUM;
 		if ((m->m_pkthdr.csum_flags & CSUM_TCP) != 0)
 			cflags |= TD_TCPCSUM;
 		if ((m->m_pkthdr.csum_flags & CSUM_UDP) != 0)
 			cflags |= TD_UDPCSUM;
 		/* Set TCP/UDP header offset. */
 		cflags |= (poff << TD_L4HDR_OFFSET_SHIFT) &
 		    TD_L4HDR_OFFSET_MASK;
 #endif
 	}
 	for (; idx < nsegs; idx++) {
 		desc = &sc->alc_rdata.alc_tx_ring[prod];
 		desc->len = htole32(TX_BYTES(txsegs[idx].ds_len) | vtag);
 		desc->flags = htole32(cflags);
 		desc->addr = htole64(txsegs[idx].ds_addr);
 		sc->alc_cdata.alc_tx_cnt++;
 		ALC_DESC_INC(prod, ALC_TX_RING_CNT);
 	}
 	/* Update producer index. */
 	sc->alc_cdata.alc_tx_prod = prod;
 
 	/* Finally set EOP on the last descriptor. */
 	prod = (prod + ALC_TX_RING_CNT - 1) % ALC_TX_RING_CNT;
 	desc = &sc->alc_rdata.alc_tx_ring[prod];
 	desc->flags |= htole32(TD_EOP);
 
 	/* Swap dmamap of the first and the last. */
 	txd = &sc->alc_cdata.alc_txdesc[prod];
 	map = txd_last->tx_dmamap;
 	txd_last->tx_dmamap = txd->tx_dmamap;
 	txd->tx_dmamap = map;
 	txd->tx_m = m;
 
 	return (0);
 }
 
 static void
 alc_start(struct ifnet *ifp)
 {
 	struct alc_softc *sc;
 
 	sc = ifp->if_softc;
 	ALC_LOCK(sc);
 	alc_start_locked(ifp);
 	ALC_UNLOCK(sc);
 }
 
 static void
 alc_start_locked(struct ifnet *ifp)
 {
 	struct alc_softc *sc;
 	struct mbuf *m_head;
 	int enq;
 
 	sc = ifp->if_softc;
 
 	ALC_LOCK_ASSERT(sc);
 
 	/* Reclaim transmitted frames. */
 	if (sc->alc_cdata.alc_tx_cnt >= ALC_TX_DESC_HIWAT)
 		alc_txeof(sc);
 
 	if ((ifp->if_drv_flags & (IFF_DRV_RUNNING | IFF_DRV_OACTIVE)) !=
 	    IFF_DRV_RUNNING || (sc->alc_flags & ALC_FLAG_LINK) == 0)
 		return;
 
 	for (enq = 0; !IFQ_DRV_IS_EMPTY(&ifp->if_snd); ) {
 		IFQ_DRV_DEQUEUE(&ifp->if_snd, m_head);
 		if (m_head == NULL)
 			break;
 		/*
 		 * Pack the data into the transmit ring. If we
 		 * don't have room, set the OACTIVE flag and wait
 		 * for the NIC to drain the ring.
 		 */
 		if (alc_encap(sc, &m_head)) {
 			if (m_head == NULL)
 				break;
 			IFQ_DRV_PREPEND(&ifp->if_snd, m_head);
 			ifp->if_drv_flags |= IFF_DRV_OACTIVE;
 			break;
 		}
 
 		enq++;
 		/*
 		 * If there's a BPF listener, bounce a copy of this frame
 		 * to him.
 		 */
 		ETHER_BPF_MTAP(ifp, m_head);
 	}
 
 	if (enq > 0) {
 		/* Sync descriptors. */
 		bus_dmamap_sync(sc->alc_cdata.alc_tx_ring_tag,
 		    sc->alc_cdata.alc_tx_ring_map, BUS_DMASYNC_PREWRITE);
 		/* Kick. Assume we're using normal Tx priority queue. */
 		if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 			CSR_WRITE_2(sc, ALC_MBOX_TD_PRI0_PROD_IDX,
 			    (uint16_t)sc->alc_cdata.alc_tx_prod);
 		else
 			CSR_WRITE_4(sc, ALC_MBOX_TD_PROD_IDX,
 			    (sc->alc_cdata.alc_tx_prod <<
 			    MBOX_TD_PROD_LO_IDX_SHIFT) &
 			    MBOX_TD_PROD_LO_IDX_MASK);
 		/* Set a timeout in case the chip goes out to lunch. */
 		sc->alc_watchdog_timer = ALC_TX_TIMEOUT;
 	}
 }
 
 static void
 alc_watchdog(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 
 	ALC_LOCK_ASSERT(sc);
 
 	if (sc->alc_watchdog_timer == 0 || --sc->alc_watchdog_timer)
 		return;
 
 	ifp = sc->alc_ifp;
 	if ((sc->alc_flags & ALC_FLAG_LINK) == 0) {
 		if_printf(sc->alc_ifp, "watchdog timeout (lost link)\n");
 		if_inc_counter(ifp, IFCOUNTER_OERRORS, 1);
 		ifp->if_drv_flags &= ~IFF_DRV_RUNNING;
 		alc_init_locked(sc);
 		return;
 	}
 	if_printf(sc->alc_ifp, "watchdog timeout -- resetting\n");
 	if_inc_counter(ifp, IFCOUNTER_OERRORS, 1);
 	ifp->if_drv_flags &= ~IFF_DRV_RUNNING;
 	alc_init_locked(sc);
 	if (!IFQ_DRV_IS_EMPTY(&ifp->if_snd))
 		alc_start_locked(ifp);
 }
 
 static int
 alc_ioctl(struct ifnet *ifp, u_long cmd, caddr_t data)
 {
 	struct alc_softc *sc;
 	struct ifreq *ifr;
 	struct mii_data *mii;
 	int error, mask;
 
 	sc = ifp->if_softc;
 	ifr = (struct ifreq *)data;
 	error = 0;
 	switch (cmd) {
 	case SIOCSIFMTU:
 		if (ifr->ifr_mtu < ETHERMIN ||
 		    ifr->ifr_mtu > (sc->alc_ident->max_framelen -
 		    sizeof(struct ether_vlan_header) - ETHER_CRC_LEN) ||
 		    ((sc->alc_flags & ALC_FLAG_JUMBO) == 0 &&
 		    ifr->ifr_mtu > ETHERMTU))
 			error = EINVAL;
 		else if (ifp->if_mtu != ifr->ifr_mtu) {
 			ALC_LOCK(sc);
 			ifp->if_mtu = ifr->ifr_mtu;
 			/* AR81[3567]x has 13 bits MSS field. */
 			if (ifp->if_mtu > ALC_TSO_MTU &&
 			    (ifp->if_capenable & IFCAP_TSO4) != 0) {
 				ifp->if_capenable &= ~IFCAP_TSO4;
 				ifp->if_hwassist &= ~CSUM_TSO;
 				VLAN_CAPABILITIES(ifp);
 			}
 			ALC_UNLOCK(sc);
 		}
 		break;
 	case SIOCSIFFLAGS:
 		ALC_LOCK(sc);
 		if ((ifp->if_flags & IFF_UP) != 0) {
 			if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0 &&
 			    ((ifp->if_flags ^ sc->alc_if_flags) &
 			    (IFF_PROMISC | IFF_ALLMULTI)) != 0)
 				alc_rxfilter(sc);
 			else
 				alc_init_locked(sc);
 		} else if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0)
 			alc_stop(sc);
 		sc->alc_if_flags = ifp->if_flags;
 		ALC_UNLOCK(sc);
 		break;
 	case SIOCADDMULTI:
 	case SIOCDELMULTI:
 		ALC_LOCK(sc);
 		if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0)
 			alc_rxfilter(sc);
 		ALC_UNLOCK(sc);
 		break;
 	case SIOCSIFMEDIA:
 	case SIOCGIFMEDIA:
 		mii = device_get_softc(sc->alc_miibus);
 		error = ifmedia_ioctl(ifp, ifr, &mii->mii_media, cmd);
 		break;
 	case SIOCSIFCAP:
 		ALC_LOCK(sc);
 		mask = ifr->ifr_reqcap ^ ifp->if_capenable;
 		if ((mask & IFCAP_TXCSUM) != 0 &&
 		    (ifp->if_capabilities & IFCAP_TXCSUM) != 0) {
 			ifp->if_capenable ^= IFCAP_TXCSUM;
 			if ((ifp->if_capenable & IFCAP_TXCSUM) != 0)
 				ifp->if_hwassist |= ALC_CSUM_FEATURES;
 			else
 				ifp->if_hwassist &= ~ALC_CSUM_FEATURES;
 		}
 		if ((mask & IFCAP_TSO4) != 0 &&
 		    (ifp->if_capabilities & IFCAP_TSO4) != 0) {
 			ifp->if_capenable ^= IFCAP_TSO4;
 			if ((ifp->if_capenable & IFCAP_TSO4) != 0) {
 				/* AR81[3567]x has 13 bits MSS field. */
 				if (ifp->if_mtu > ALC_TSO_MTU) {
 					ifp->if_capenable &= ~IFCAP_TSO4;
 					ifp->if_hwassist &= ~CSUM_TSO;
 				} else
 					ifp->if_hwassist |= CSUM_TSO;
 			} else
 				ifp->if_hwassist &= ~CSUM_TSO;
 		}
 		if ((mask & IFCAP_WOL_MCAST) != 0 &&
 		    (ifp->if_capabilities & IFCAP_WOL_MCAST) != 0)
 			ifp->if_capenable ^= IFCAP_WOL_MCAST;
 		if ((mask & IFCAP_WOL_MAGIC) != 0 &&
 		    (ifp->if_capabilities & IFCAP_WOL_MAGIC) != 0)
 			ifp->if_capenable ^= IFCAP_WOL_MAGIC;
 		if ((mask & IFCAP_VLAN_HWTAGGING) != 0 &&
 		    (ifp->if_capabilities & IFCAP_VLAN_HWTAGGING) != 0) {
 			ifp->if_capenable ^= IFCAP_VLAN_HWTAGGING;
 			alc_rxvlan(sc);
 		}
 		if ((mask & IFCAP_VLAN_HWCSUM) != 0 &&
 		    (ifp->if_capabilities & IFCAP_VLAN_HWCSUM) != 0)
 			ifp->if_capenable ^= IFCAP_VLAN_HWCSUM;
 		if ((mask & IFCAP_VLAN_HWTSO) != 0 &&
 		    (ifp->if_capabilities & IFCAP_VLAN_HWTSO) != 0)
 			ifp->if_capenable ^= IFCAP_VLAN_HWTSO;
 		if ((ifp->if_capenable & IFCAP_VLAN_HWTAGGING) == 0)
 			ifp->if_capenable &=
 			    ~(IFCAP_VLAN_HWTSO | IFCAP_VLAN_HWCSUM);
 		ALC_UNLOCK(sc);
 		VLAN_CAPABILITIES(ifp);
 		break;
 	default:
 		error = ether_ioctl(ifp, cmd, data);
 		break;
 	}
 
 	return (error);
 }
 
 static void
 alc_mac_config(struct alc_softc *sc)
 {
 	struct mii_data *mii;
 	uint32_t reg;
 
 	ALC_LOCK_ASSERT(sc);
 
 	mii = device_get_softc(sc->alc_miibus);
 	reg = CSR_READ_4(sc, ALC_MAC_CFG);
 	reg &= ~(MAC_CFG_FULL_DUPLEX | MAC_CFG_TX_FC | MAC_CFG_RX_FC |
 	    MAC_CFG_SPEED_MASK);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151_V2 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B2)
 		reg |= MAC_CFG_HASH_ALG_CRC32 | MAC_CFG_SPEED_MODE_SW;
 	/* Reprogram MAC with resolved speed/duplex. */
 	switch (IFM_SUBTYPE(mii->mii_media_active)) {
 	case IFM_10_T:
 	case IFM_100_TX:
 		reg |= MAC_CFG_SPEED_10_100;
 		break;
 	case IFM_1000_T:
 		reg |= MAC_CFG_SPEED_1000;
 		break;
 	}
 	if ((IFM_OPTIONS(mii->mii_media_active) & IFM_FDX) != 0) {
 		reg |= MAC_CFG_FULL_DUPLEX;
 		if ((IFM_OPTIONS(mii->mii_media_active) & IFM_ETH_TXPAUSE) != 0)
 			reg |= MAC_CFG_TX_FC;
 		if ((IFM_OPTIONS(mii->mii_media_active) & IFM_ETH_RXPAUSE) != 0)
 			reg |= MAC_CFG_RX_FC;
 	}
 	CSR_WRITE_4(sc, ALC_MAC_CFG, reg);
 }
 
 static void
 alc_stats_clear(struct alc_softc *sc)
 {
 	struct smb sb, *smb;
 	uint32_t *reg;
 	int i;
 
 	if ((sc->alc_flags & ALC_FLAG_SMB_BUG) == 0) {
 		bus_dmamap_sync(sc->alc_cdata.alc_smb_tag,
 		    sc->alc_cdata.alc_smb_map,
 		    BUS_DMASYNC_POSTREAD | BUS_DMASYNC_POSTWRITE);
 		smb = sc->alc_rdata.alc_smb;
 		/* Update done, clear. */
 		smb->updated = 0;
 		bus_dmamap_sync(sc->alc_cdata.alc_smb_tag,
 		    sc->alc_cdata.alc_smb_map,
 		    BUS_DMASYNC_PREREAD | BUS_DMASYNC_PREWRITE);
 	} else {
 		for (reg = &sb.rx_frames, i = 0; reg <= &sb.rx_pkts_filtered;
 		    reg++) {
 			CSR_READ_4(sc, ALC_RX_MIB_BASE + i);
 			i += sizeof(uint32_t);
 		}
 		/* Read Tx statistics. */
 		for (reg = &sb.tx_frames, i = 0; reg <= &sb.tx_mcast_bytes;
 		    reg++) {
 			CSR_READ_4(sc, ALC_TX_MIB_BASE + i);
 			i += sizeof(uint32_t);
 		}
 	}
 }
 
 static void
 alc_stats_update(struct alc_softc *sc)
 {
 	struct alc_hw_stats *stat;
 	struct smb sb, *smb;
 	struct ifnet *ifp;
 	uint32_t *reg;
 	int i;
 
 	ALC_LOCK_ASSERT(sc);
 
 	ifp = sc->alc_ifp;
 	stat = &sc->alc_stats;
 	if ((sc->alc_flags & ALC_FLAG_SMB_BUG) == 0) {
 		bus_dmamap_sync(sc->alc_cdata.alc_smb_tag,
 		    sc->alc_cdata.alc_smb_map,
 		    BUS_DMASYNC_POSTREAD | BUS_DMASYNC_POSTWRITE);
 		smb = sc->alc_rdata.alc_smb;
 		if (smb->updated == 0)
 			return;
 	} else {
 		smb = &sb;
 		/* Read Rx statistics. */
 		for (reg = &sb.rx_frames, i = 0; reg <= &sb.rx_pkts_filtered;
 		    reg++) {
 			*reg = CSR_READ_4(sc, ALC_RX_MIB_BASE + i);
 			i += sizeof(uint32_t);
 		}
 		/* Read Tx statistics. */
 		for (reg = &sb.tx_frames, i = 0; reg <= &sb.tx_mcast_bytes;
 		    reg++) {
 			*reg = CSR_READ_4(sc, ALC_TX_MIB_BASE + i);
 			i += sizeof(uint32_t);
 		}
 	}
 
 	/* Rx stats. */
 	stat->rx_frames += smb->rx_frames;
 	stat->rx_bcast_frames += smb->rx_bcast_frames;
 	stat->rx_mcast_frames += smb->rx_mcast_frames;
 	stat->rx_pause_frames += smb->rx_pause_frames;
 	stat->rx_control_frames += smb->rx_control_frames;
 	stat->rx_crcerrs += smb->rx_crcerrs;
 	stat->rx_lenerrs += smb->rx_lenerrs;
 	stat->rx_bytes += smb->rx_bytes;
 	stat->rx_runts += smb->rx_runts;
 	stat->rx_fragments += smb->rx_fragments;
 	stat->rx_pkts_64 += smb->rx_pkts_64;
 	stat->rx_pkts_65_127 += smb->rx_pkts_65_127;
 	stat->rx_pkts_128_255 += smb->rx_pkts_128_255;
 	stat->rx_pkts_256_511 += smb->rx_pkts_256_511;
 	stat->rx_pkts_512_1023 += smb->rx_pkts_512_1023;
 	stat->rx_pkts_1024_1518 += smb->rx_pkts_1024_1518;
 	stat->rx_pkts_1519_max += smb->rx_pkts_1519_max;
 	stat->rx_pkts_truncated += smb->rx_pkts_truncated;
 	stat->rx_fifo_oflows += smb->rx_fifo_oflows;
 	stat->rx_rrs_errs += smb->rx_rrs_errs;
 	stat->rx_alignerrs += smb->rx_alignerrs;
 	stat->rx_bcast_bytes += smb->rx_bcast_bytes;
 	stat->rx_mcast_bytes += smb->rx_mcast_bytes;
 	stat->rx_pkts_filtered += smb->rx_pkts_filtered;
 
 	/* Tx stats. */
 	stat->tx_frames += smb->tx_frames;
 	stat->tx_bcast_frames += smb->tx_bcast_frames;
 	stat->tx_mcast_frames += smb->tx_mcast_frames;
 	stat->tx_pause_frames += smb->tx_pause_frames;
 	stat->tx_excess_defer += smb->tx_excess_defer;
 	stat->tx_control_frames += smb->tx_control_frames;
 	stat->tx_deferred += smb->tx_deferred;
 	stat->tx_bytes += smb->tx_bytes;
 	stat->tx_pkts_64 += smb->tx_pkts_64;
 	stat->tx_pkts_65_127 += smb->tx_pkts_65_127;
 	stat->tx_pkts_128_255 += smb->tx_pkts_128_255;
 	stat->tx_pkts_256_511 += smb->tx_pkts_256_511;
 	stat->tx_pkts_512_1023 += smb->tx_pkts_512_1023;
 	stat->tx_pkts_1024_1518 += smb->tx_pkts_1024_1518;
 	stat->tx_pkts_1519_max += smb->tx_pkts_1519_max;
 	stat->tx_single_colls += smb->tx_single_colls;
 	stat->tx_multi_colls += smb->tx_multi_colls;
 	stat->tx_late_colls += smb->tx_late_colls;
 	stat->tx_excess_colls += smb->tx_excess_colls;
 	stat->tx_underrun += smb->tx_underrun;
 	stat->tx_desc_underrun += smb->tx_desc_underrun;
 	stat->tx_lenerrs += smb->tx_lenerrs;
 	stat->tx_pkts_truncated += smb->tx_pkts_truncated;
 	stat->tx_bcast_bytes += smb->tx_bcast_bytes;
 	stat->tx_mcast_bytes += smb->tx_mcast_bytes;
 
 	/* Update counters in ifnet. */
 	if_inc_counter(ifp, IFCOUNTER_OPACKETS, smb->tx_frames);
 
 	if_inc_counter(ifp, IFCOUNTER_COLLISIONS, smb->tx_single_colls +
 	    smb->tx_multi_colls * 2 + smb->tx_late_colls +
 	    smb->tx_excess_colls * HDPX_CFG_RETRY_DEFAULT);
 
 	if_inc_counter(ifp, IFCOUNTER_OERRORS, smb->tx_late_colls +
 	    smb->tx_excess_colls + smb->tx_underrun + smb->tx_pkts_truncated);
 
 	if_inc_counter(ifp, IFCOUNTER_IPACKETS, smb->rx_frames);
 
 	if_inc_counter(ifp, IFCOUNTER_IERRORS,
 	    smb->rx_crcerrs + smb->rx_lenerrs +
 	    smb->rx_runts + smb->rx_pkts_truncated +
 	    smb->rx_fifo_oflows + smb->rx_rrs_errs +
 	    smb->rx_alignerrs);
 
 	if ((sc->alc_flags & ALC_FLAG_SMB_BUG) == 0) {
 		/* Update done, clear. */
 		smb->updated = 0;
 		bus_dmamap_sync(sc->alc_cdata.alc_smb_tag,
 		    sc->alc_cdata.alc_smb_map,
 		    BUS_DMASYNC_PREREAD | BUS_DMASYNC_PREWRITE);
 	}
 }
 
 static int
 alc_intr(void *arg)
 {
 	struct alc_softc *sc;
 	uint32_t status;
 
 	sc = (struct alc_softc *)arg;
 
 	status = CSR_READ_4(sc, ALC_INTR_STATUS);
 	if ((status & ALC_INTRS) == 0)
 		return (FILTER_STRAY);
 	/* Disable interrupts. */
 	CSR_WRITE_4(sc, ALC_INTR_STATUS, INTR_DIS_INT);
 	taskqueue_enqueue(sc->alc_tq, &sc->alc_int_task);
 
 	return (FILTER_HANDLED);
 }
 
 static void
 alc_int_task(void *arg, int pending)
 {
 	struct alc_softc *sc;
 	struct ifnet *ifp;
 	uint32_t status;
 	int more;
 
 	sc = (struct alc_softc *)arg;
 	ifp = sc->alc_ifp;
 
 	status = CSR_READ_4(sc, ALC_INTR_STATUS);
 	ALC_LOCK(sc);
 	if (sc->alc_morework != 0) {
 		sc->alc_morework = 0;
 		status |= INTR_RX_PKT;
 	}
 	if ((status & ALC_INTRS) == 0)
 		goto done;
 
 	/* Acknowledge interrupts but still disable interrupts. */
 	CSR_WRITE_4(sc, ALC_INTR_STATUS, status | INTR_DIS_INT);
 
 	more = 0;
 	if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0) {
 		if ((status & INTR_RX_PKT) != 0) {
 			more = alc_rxintr(sc, sc->alc_process_limit);
 			if (more == EAGAIN)
 				sc->alc_morework = 1;
 			else if (more == EIO) {
 				ifp->if_drv_flags &= ~IFF_DRV_RUNNING;
 				alc_init_locked(sc);
 				ALC_UNLOCK(sc);
 				return;
 			}
 		}
 		if ((status & (INTR_DMA_RD_TO_RST | INTR_DMA_WR_TO_RST |
 		    INTR_TXQ_TO_RST)) != 0) {
 			if ((status & INTR_DMA_RD_TO_RST) != 0)
 				device_printf(sc->alc_dev,
 				    "DMA read error! -- resetting\n");
 			if ((status & INTR_DMA_WR_TO_RST) != 0)
 				device_printf(sc->alc_dev,
 				    "DMA write error! -- resetting\n");
 			if ((status & INTR_TXQ_TO_RST) != 0)
 				device_printf(sc->alc_dev,
 				    "TxQ reset! -- resetting\n");
 			ifp->if_drv_flags &= ~IFF_DRV_RUNNING;
 			alc_init_locked(sc);
 			ALC_UNLOCK(sc);
 			return;
 		}
 		if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0 &&
 		    !IFQ_DRV_IS_EMPTY(&ifp->if_snd))
 			alc_start_locked(ifp);
 	}
 
 	if (more == EAGAIN ||
 	    (CSR_READ_4(sc, ALC_INTR_STATUS) & ALC_INTRS) != 0) {
 		ALC_UNLOCK(sc);
 		taskqueue_enqueue(sc->alc_tq, &sc->alc_int_task);
 		return;
 	}
 
 done:
 	if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0) {
 		/* Re-enable interrupts if we're running. */
 		CSR_WRITE_4(sc, ALC_INTR_STATUS, 0x7FFFFFFF);
 	}
 	ALC_UNLOCK(sc);
 }
 
 static void
 alc_txeof(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	struct alc_txdesc *txd;
 	uint32_t cons, prod;
 	int prog;
 
 	ALC_LOCK_ASSERT(sc);
 
 	ifp = sc->alc_ifp;
 
 	if (sc->alc_cdata.alc_tx_cnt == 0)
 		return;
 	bus_dmamap_sync(sc->alc_cdata.alc_tx_ring_tag,
 	    sc->alc_cdata.alc_tx_ring_map, BUS_DMASYNC_POSTWRITE);
 	if ((sc->alc_flags & ALC_FLAG_CMB_BUG) == 0) {
 		bus_dmamap_sync(sc->alc_cdata.alc_cmb_tag,
 		    sc->alc_cdata.alc_cmb_map, BUS_DMASYNC_POSTREAD);
 		prod = sc->alc_rdata.alc_cmb->cons;
 	} else {
 		if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 			prod = CSR_READ_2(sc, ALC_MBOX_TD_PRI0_CONS_IDX);
 		else {
 			prod = CSR_READ_4(sc, ALC_MBOX_TD_CONS_IDX);
 			/* Assume we're using normal Tx priority queue. */
 			prod = (prod & MBOX_TD_CONS_LO_IDX_MASK) >>
 			    MBOX_TD_CONS_LO_IDX_SHIFT;
 		}
 	}
 	cons = sc->alc_cdata.alc_tx_cons;
 	/*
 	 * Go through our Tx list and free mbufs for those
 	 * frames which have been transmitted.
 	 */
 	for (prog = 0; cons != prod; prog++,
 	    ALC_DESC_INC(cons, ALC_TX_RING_CNT)) {
 		if (sc->alc_cdata.alc_tx_cnt <= 0)
 			break;
 		prog++;
 		ifp->if_drv_flags &= ~IFF_DRV_OACTIVE;
 		sc->alc_cdata.alc_tx_cnt--;
 		txd = &sc->alc_cdata.alc_txdesc[cons];
 		if (txd->tx_m != NULL) {
 			/* Reclaim transmitted mbufs. */
 			bus_dmamap_sync(sc->alc_cdata.alc_tx_tag,
 			    txd->tx_dmamap, BUS_DMASYNC_POSTWRITE);
 			bus_dmamap_unload(sc->alc_cdata.alc_tx_tag,
 			    txd->tx_dmamap);
 			m_freem(txd->tx_m);
 			txd->tx_m = NULL;
 		}
 	}
 
 	if ((sc->alc_flags & ALC_FLAG_CMB_BUG) == 0)
 		bus_dmamap_sync(sc->alc_cdata.alc_cmb_tag,
 		    sc->alc_cdata.alc_cmb_map, BUS_DMASYNC_PREREAD);
 	sc->alc_cdata.alc_tx_cons = cons;
 	/*
 	 * Unarm watchdog timer only when there is no pending
 	 * frames in Tx queue.
 	 */
 	if (sc->alc_cdata.alc_tx_cnt == 0)
 		sc->alc_watchdog_timer = 0;
 }
 
 static int
 alc_newbuf(struct alc_softc *sc, struct alc_rxdesc *rxd)
 {
 	struct mbuf *m;
 	bus_dma_segment_t segs[1];
 	bus_dmamap_t map;
 	int nsegs;
 
 	m = m_getcl(M_NOWAIT, MT_DATA, M_PKTHDR);
 	if (m == NULL)
 		return (ENOBUFS);
 	m->m_len = m->m_pkthdr.len = RX_BUF_SIZE_MAX;
 #ifndef __NO_STRICT_ALIGNMENT
 	m_adj(m, sizeof(uint64_t));
 #endif
 
 	if (bus_dmamap_load_mbuf_sg(sc->alc_cdata.alc_rx_tag,
 	    sc->alc_cdata.alc_rx_sparemap, m, segs, &nsegs, 0) != 0) {
 		m_freem(m);
 		return (ENOBUFS);
 	}
 	KASSERT(nsegs == 1, ("%s: %d segments returned!", __func__, nsegs));
 
 	if (rxd->rx_m != NULL) {
 		bus_dmamap_sync(sc->alc_cdata.alc_rx_tag, rxd->rx_dmamap,
 		    BUS_DMASYNC_POSTREAD);
 		bus_dmamap_unload(sc->alc_cdata.alc_rx_tag, rxd->rx_dmamap);
 	}
 	map = rxd->rx_dmamap;
 	rxd->rx_dmamap = sc->alc_cdata.alc_rx_sparemap;
 	sc->alc_cdata.alc_rx_sparemap = map;
 	bus_dmamap_sync(sc->alc_cdata.alc_rx_tag, rxd->rx_dmamap,
 	    BUS_DMASYNC_PREREAD);
 	rxd->rx_m = m;
 	rxd->rx_desc->addr = htole64(segs[0].ds_addr);
 	return (0);
 }
 
 static int
 alc_rxintr(struct alc_softc *sc, int count)
 {
 	struct ifnet *ifp;
 	struct rx_rdesc *rrd;
 	uint32_t nsegs, status;
 	int rr_cons, prog;
 
 	bus_dmamap_sync(sc->alc_cdata.alc_rr_ring_tag,
 	    sc->alc_cdata.alc_rr_ring_map,
 	    BUS_DMASYNC_POSTREAD | BUS_DMASYNC_POSTWRITE);
 	bus_dmamap_sync(sc->alc_cdata.alc_rx_ring_tag,
 	    sc->alc_cdata.alc_rx_ring_map, BUS_DMASYNC_POSTWRITE);
 	rr_cons = sc->alc_cdata.alc_rr_cons;
 	ifp = sc->alc_ifp;
 	for (prog = 0; (ifp->if_drv_flags & IFF_DRV_RUNNING) != 0;) {
 		if (count-- <= 0)
 			break;
 		rrd = &sc->alc_rdata.alc_rr_ring[rr_cons];
 		status = le32toh(rrd->status);
 		if ((status & RRD_VALID) == 0)
 			break;
 		nsegs = RRD_RD_CNT(le32toh(rrd->rdinfo));
 		if (nsegs == 0) {
 			/* This should not happen! */
 			device_printf(sc->alc_dev,
 			    "unexpected segment count -- resetting\n");
 			return (EIO);
 		}
 		alc_rxeof(sc, rrd);
 		/* Clear Rx return status. */
 		rrd->status = 0;
 		ALC_DESC_INC(rr_cons, ALC_RR_RING_CNT);
 		sc->alc_cdata.alc_rx_cons += nsegs;
 		sc->alc_cdata.alc_rx_cons %= ALC_RR_RING_CNT;
 		prog += nsegs;
 	}
 
 	if (prog > 0) {
 		/* Update the consumer index. */
 		sc->alc_cdata.alc_rr_cons = rr_cons;
 		/* Sync Rx return descriptors. */
 		bus_dmamap_sync(sc->alc_cdata.alc_rr_ring_tag,
 		    sc->alc_cdata.alc_rr_ring_map,
 		    BUS_DMASYNC_PREREAD | BUS_DMASYNC_PREWRITE);
 		/*
 		 * Sync updated Rx descriptors such that controller see
 		 * modified buffer addresses.
 		 */
 		bus_dmamap_sync(sc->alc_cdata.alc_rx_ring_tag,
 		    sc->alc_cdata.alc_rx_ring_map, BUS_DMASYNC_PREWRITE);
 		/*
 		 * Let controller know availability of new Rx buffers.
 		 * Since alc(4) use RXQ_CFG_RD_BURST_DEFAULT descriptors
 		 * it may be possible to update ALC_MBOX_RD0_PROD_IDX
 		 * only when Rx buffer pre-fetching is required. In
 		 * addition we already set ALC_RX_RD_FREE_THRESH to
 		 * RX_RD_FREE_THRESH_LO_DEFAULT descriptors. However
 		 * it still seems that pre-fetching needs more
 		 * experimentation.
 		 */
 		if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 			CSR_WRITE_2(sc, ALC_MBOX_RD0_PROD_IDX,
 			    (uint16_t)sc->alc_cdata.alc_rx_cons);
 		else
 			CSR_WRITE_4(sc, ALC_MBOX_RD0_PROD_IDX,
 			    sc->alc_cdata.alc_rx_cons);
 	}
 
 	return (count > 0 ? 0 : EAGAIN);
 }
 
 #ifndef __NO_STRICT_ALIGNMENT
 static struct mbuf *
 alc_fixup_rx(struct ifnet *ifp, struct mbuf *m)
 {
 	struct mbuf *n;
         int i;
         uint16_t *src, *dst;
 
 	src = mtod(m, uint16_t *);
 	dst = src - 3;
 
 	if (m->m_next == NULL) {
 		for (i = 0; i < (m->m_len / sizeof(uint16_t) + 1); i++)
 			*dst++ = *src++;
 		m->m_data -= 6;
 		return (m);
 	}
 	/*
 	 * Append a new mbuf to received mbuf chain and copy ethernet
 	 * header from the mbuf chain. This can save lots of CPU
 	 * cycles for jumbo frame.
 	 */
 	MGETHDR(n, M_NOWAIT, MT_DATA);
 	if (n == NULL) {
 		if_inc_counter(ifp, IFCOUNTER_IQDROPS, 1);
 		m_freem(m);
 		return (NULL);
 	}
 	bcopy(m->m_data, n->m_data, ETHER_HDR_LEN);
 	m->m_data += ETHER_HDR_LEN;
 	m->m_len -= ETHER_HDR_LEN;
 	n->m_len = ETHER_HDR_LEN;
 	M_MOVE_PKTHDR(n, m);
 	n->m_next = m;
 	return (n);
 }
 #endif
 
 /* Receive a frame. */
 static void
 alc_rxeof(struct alc_softc *sc, struct rx_rdesc *rrd)
 {
 	struct alc_rxdesc *rxd;
 	struct ifnet *ifp;
 	struct mbuf *mp, *m;
 	uint32_t rdinfo, status, vtag;
 	int count, nsegs, rx_cons;
 
 	ifp = sc->alc_ifp;
 	status = le32toh(rrd->status);
 	rdinfo = le32toh(rrd->rdinfo);
 	rx_cons = RRD_RD_IDX(rdinfo);
 	nsegs = RRD_RD_CNT(rdinfo);
 
 	sc->alc_cdata.alc_rxlen = RRD_BYTES(status);
 	if ((status & (RRD_ERR_SUM | RRD_ERR_LENGTH)) != 0) {
 		/*
 		 * We want to pass the following frames to upper
 		 * layer regardless of error status of Rx return
 		 * ring.
 		 *
 		 *  o IP/TCP/UDP checksum is bad.
 		 *  o frame length and protocol specific length
 		 *     does not match.
 		 *
 		 *  Force network stack compute checksum for
 		 *  errored frames.
 		 */
 		status |= RRD_TCP_UDPCSUM_NOK | RRD_IPCSUM_NOK;
 		if ((status & (RRD_ERR_CRC | RRD_ERR_ALIGN |
 		    RRD_ERR_TRUNC | RRD_ERR_RUNT)) != 0)
 			return;
 	}
 
 	for (count = 0; count < nsegs; count++,
 	    ALC_DESC_INC(rx_cons, ALC_RX_RING_CNT)) {
 		rxd = &sc->alc_cdata.alc_rxdesc[rx_cons];
 		mp = rxd->rx_m;
 		/* Add a new receive buffer to the ring. */
 		if (alc_newbuf(sc, rxd) != 0) {
 			if_inc_counter(ifp, IFCOUNTER_IQDROPS, 1);
 			/* Reuse Rx buffers. */
 			if (sc->alc_cdata.alc_rxhead != NULL)
 				m_freem(sc->alc_cdata.alc_rxhead);
 			break;
 		}
 
 		/*
 		 * Assume we've received a full sized frame.
 		 * Actual size is fixed when we encounter the end of
 		 * multi-segmented frame.
 		 */
 		mp->m_len = sc->alc_buf_size;
 
 		/* Chain received mbufs. */
 		if (sc->alc_cdata.alc_rxhead == NULL) {
 			sc->alc_cdata.alc_rxhead = mp;
 			sc->alc_cdata.alc_rxtail = mp;
 		} else {
 			mp->m_flags &= ~M_PKTHDR;
 			sc->alc_cdata.alc_rxprev_tail =
 			    sc->alc_cdata.alc_rxtail;
 			sc->alc_cdata.alc_rxtail->m_next = mp;
 			sc->alc_cdata.alc_rxtail = mp;
 		}
 
 		if (count == nsegs - 1) {
 			/* Last desc. for this frame. */
 			m = sc->alc_cdata.alc_rxhead;
 			m->m_flags |= M_PKTHDR;
 			/*
 			 * It seems that L1C/L2C controller has no way
 			 * to tell hardware to strip CRC bytes.
 			 */
 			m->m_pkthdr.len =
 			    sc->alc_cdata.alc_rxlen - ETHER_CRC_LEN;
 			if (nsegs > 1) {
 				/* Set last mbuf size. */
 				mp->m_len = sc->alc_cdata.alc_rxlen -
 				    (nsegs - 1) * sc->alc_buf_size;
 				/* Remove the CRC bytes in chained mbufs. */
 				if (mp->m_len <= ETHER_CRC_LEN) {
 					sc->alc_cdata.alc_rxtail =
 					    sc->alc_cdata.alc_rxprev_tail;
 					sc->alc_cdata.alc_rxtail->m_len -=
 					    (ETHER_CRC_LEN - mp->m_len);
 					sc->alc_cdata.alc_rxtail->m_next = NULL;
 					m_freem(mp);
 				} else {
 					mp->m_len -= ETHER_CRC_LEN;
 				}
 			} else
 				m->m_len = m->m_pkthdr.len;
 			m->m_pkthdr.rcvif = ifp;
 			/*
 			 * Due to hardware bugs, Rx checksum offloading
 			 * was intentionally disabled.
 			 */
 			if ((ifp->if_capenable & IFCAP_VLAN_HWTAGGING) != 0 &&
 			    (status & RRD_VLAN_TAG) != 0) {
 				vtag = RRD_VLAN(le32toh(rrd->vtag));
 				m->m_pkthdr.ether_vtag = ntohs(vtag);
 				m->m_flags |= M_VLANTAG;
 			}
 #ifndef __NO_STRICT_ALIGNMENT
 			m = alc_fixup_rx(ifp, m);
 			if (m != NULL)
 #endif
 			{
 			/* Pass it on. */
 			ALC_UNLOCK(sc);
 			(*ifp->if_input)(ifp, m);
 			ALC_LOCK(sc);
 			}
 		}
 	}
 	/* Reset mbuf chains. */
 	ALC_RXCHAIN_RESET(sc);
 }
 
 static void
 alc_tick(void *arg)
 {
 	struct alc_softc *sc;
 	struct mii_data *mii;
 
 	sc = (struct alc_softc *)arg;
 
 	ALC_LOCK_ASSERT(sc);
 
 	mii = device_get_softc(sc->alc_miibus);
 	mii_tick(mii);
 	alc_stats_update(sc);
 	/*
 	 * alc(4) does not rely on Tx completion interrupts to reclaim
 	 * transferred buffers. Instead Tx completion interrupts are
 	 * used to hint for scheduling Tx task. So it's necessary to
 	 * release transmitted buffers by kicking Tx completion
 	 * handler. This limits the maximum reclamation delay to a hz.
 	 */
 	alc_txeof(sc);
 	alc_watchdog(sc);
 	callout_reset(&sc->alc_tick_ch, hz, alc_tick, sc);
 }
 
 static void
 alc_osc_reset(struct alc_softc *sc)
 {
 	uint32_t reg;
 
 	reg = CSR_READ_4(sc, ALC_MISC3);
 	reg &= ~MISC3_25M_BY_SW;
 	reg |= MISC3_25M_NOTO_INTNL;
 	CSR_WRITE_4(sc, ALC_MISC3, reg);
 
 	reg = CSR_READ_4(sc, ALC_MISC);
 	if (AR816X_REV(sc->alc_rev) >= AR816X_REV_B0) {
 		/*
 		 * Restore over-current protection default value.
 		 * This value could be reset by MAC reset.
 		 */
 		reg &= ~MISC_PSW_OCP_MASK;
 		reg |= (MISC_PSW_OCP_DEFAULT << MISC_PSW_OCP_SHIFT);
 		reg &= ~MISC_INTNLOSC_OPEN;
 		CSR_WRITE_4(sc, ALC_MISC, reg);
 		CSR_WRITE_4(sc, ALC_MISC, reg | MISC_INTNLOSC_OPEN);
 		reg = CSR_READ_4(sc, ALC_MISC2);
 		reg &= ~MISC2_CALB_START;
 		CSR_WRITE_4(sc, ALC_MISC2, reg);
 		CSR_WRITE_4(sc, ALC_MISC2, reg | MISC2_CALB_START);
 
 	} else {
 		reg &= ~MISC_INTNLOSC_OPEN;
 		/* Disable isolate for revision A devices. */
 		if (AR816X_REV(sc->alc_rev) <= AR816X_REV_A1)
 			reg &= ~MISC_ISO_ENB;
 		CSR_WRITE_4(sc, ALC_MISC, reg | MISC_INTNLOSC_OPEN);
 		CSR_WRITE_4(sc, ALC_MISC, reg);
 	}
 
 	DELAY(20);
 }
 
 static void
 alc_reset(struct alc_softc *sc)
 {
 	uint32_t pmcfg, reg;
 	int i;
 
 	pmcfg = 0;
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		/* Reset workaround. */
 		CSR_WRITE_4(sc, ALC_MBOX_RD0_PROD_IDX, 1);
 		if (AR816X_REV(sc->alc_rev) <= AR816X_REV_A1 &&
 		    (sc->alc_rev & 0x01) != 0) {
 			/* Disable L0s/L1s before reset. */
 			pmcfg = CSR_READ_4(sc, ALC_PM_CFG);
 			if ((pmcfg & (PM_CFG_ASPM_L0S_ENB | PM_CFG_ASPM_L1_ENB))
 			    != 0) {
 				pmcfg &= ~(PM_CFG_ASPM_L0S_ENB |
 				    PM_CFG_ASPM_L1_ENB);
 				CSR_WRITE_4(sc, ALC_PM_CFG, pmcfg);
 			}
 		}
 	}
 	reg = CSR_READ_4(sc, ALC_MASTER_CFG);
 	reg |= MASTER_OOB_DIS_OFF | MASTER_RESET;
 	CSR_WRITE_4(sc, ALC_MASTER_CFG, reg);
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		for (i = ALC_RESET_TIMEOUT; i > 0; i--) {
 			DELAY(10);
 			if (CSR_READ_4(sc, ALC_MBOX_RD0_PROD_IDX) == 0)
 				break;
 		}
 		if (i == 0)
 			device_printf(sc->alc_dev, "MAC reset timeout!\n");
 	}
 	for (i = ALC_RESET_TIMEOUT; i > 0; i--) {
 		DELAY(10);
 		if ((CSR_READ_4(sc, ALC_MASTER_CFG) & MASTER_RESET) == 0)
 			break;
 	}
 	if (i == 0)
 		device_printf(sc->alc_dev, "master reset timeout!\n");
 
 	for (i = ALC_RESET_TIMEOUT; i > 0; i--) {
 		reg = CSR_READ_4(sc, ALC_IDLE_STATUS);
 		if ((reg & (IDLE_STATUS_RXMAC | IDLE_STATUS_TXMAC |
 		    IDLE_STATUS_RXQ | IDLE_STATUS_TXQ)) == 0)
 			break;
 		DELAY(10);
 	}
 	if (i == 0)
 		device_printf(sc->alc_dev, "reset timeout(0x%08x)!\n", reg);
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		if (AR816X_REV(sc->alc_rev) <= AR816X_REV_A1 &&
 		    (sc->alc_rev & 0x01) != 0) {
 			reg = CSR_READ_4(sc, ALC_MASTER_CFG);
 			reg |= MASTER_CLK_SEL_DIS;
 			CSR_WRITE_4(sc, ALC_MASTER_CFG, reg);
 			/* Restore L0s/L1s config. */
 			if ((pmcfg & (PM_CFG_ASPM_L0S_ENB | PM_CFG_ASPM_L1_ENB))
 			    != 0)
 				CSR_WRITE_4(sc, ALC_PM_CFG, pmcfg);
 		}
 
 		alc_osc_reset(sc);
 		reg = CSR_READ_4(sc, ALC_MISC3);
 		reg &= ~MISC3_25M_BY_SW;
 		reg |= MISC3_25M_NOTO_INTNL;
 		CSR_WRITE_4(sc, ALC_MISC3, reg);
 		reg = CSR_READ_4(sc, ALC_MISC);
 		reg &= ~MISC_INTNLOSC_OPEN;
 		if (AR816X_REV(sc->alc_rev) <= AR816X_REV_A1)
 			reg &= ~MISC_ISO_ENB;
 		CSR_WRITE_4(sc, ALC_MISC, reg);
 		DELAY(20);
 	}
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151_V2)
 		CSR_WRITE_4(sc, ALC_SERDES_LOCK,
 		    CSR_READ_4(sc, ALC_SERDES_LOCK) | SERDES_MAC_CLK_SLOWDOWN |
 		    SERDES_PHY_CLK_SLOWDOWN);
 }
 
 static void
 alc_init(void *xsc)
 {
 	struct alc_softc *sc;
 
 	sc = (struct alc_softc *)xsc;
 	ALC_LOCK(sc);
 	alc_init_locked(sc);
 	ALC_UNLOCK(sc);
 }
 
 static void
 alc_init_locked(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	struct mii_data *mii;
 	uint8_t eaddr[ETHER_ADDR_LEN];
 	bus_addr_t paddr;
 	uint32_t reg, rxf_hi, rxf_lo;
 
 	ALC_LOCK_ASSERT(sc);
 
 	ifp = sc->alc_ifp;
 	mii = device_get_softc(sc->alc_miibus);
 
 	if ((ifp->if_drv_flags & IFF_DRV_RUNNING) != 0)
 		return;
 	/*
 	 * Cancel any pending I/O.
 	 */
 	alc_stop(sc);
 	/*
 	 * Reset the chip to a known state.
 	 */
 	alc_reset(sc);
 
 	/* Initialize Rx descriptors. */
 	if (alc_init_rx_ring(sc) != 0) {
 		device_printf(sc->alc_dev, "no memory for Rx buffers.\n");
 		alc_stop(sc);
 		return;
 	}
 	alc_init_rr_ring(sc);
 	alc_init_tx_ring(sc);
 	alc_init_cmb(sc);
 	alc_init_smb(sc);
 
 	/* Enable all clocks. */
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		CSR_WRITE_4(sc, ALC_CLK_GATING_CFG, CLK_GATING_DMAW_ENB |
 		    CLK_GATING_DMAR_ENB | CLK_GATING_TXQ_ENB |
 		    CLK_GATING_RXQ_ENB | CLK_GATING_TXMAC_ENB |
 		    CLK_GATING_RXMAC_ENB);
 		if (AR816X_REV(sc->alc_rev) >= AR816X_REV_B0)
 			CSR_WRITE_4(sc, ALC_IDLE_DECISN_TIMER,
 			    IDLE_DECISN_TIMER_DEFAULT_1MS);
 	} else
 		CSR_WRITE_4(sc, ALC_CLK_GATING_CFG, 0);
 
 	/* Reprogram the station address. */
 	bcopy(IF_LLADDR(ifp), eaddr, ETHER_ADDR_LEN);
 	CSR_WRITE_4(sc, ALC_PAR0,
 	    eaddr[2] << 24 | eaddr[3] << 16 | eaddr[4] << 8 | eaddr[5]);
 	CSR_WRITE_4(sc, ALC_PAR1, eaddr[0] << 8 | eaddr[1]);
 	/*
 	 * Clear WOL status and disable all WOL feature as WOL
 	 * would interfere Rx operation under normal environments.
 	 */
 	CSR_READ_4(sc, ALC_WOL_CFG);
 	CSR_WRITE_4(sc, ALC_WOL_CFG, 0);
 	/* Set Tx descriptor base addresses. */
 	paddr = sc->alc_rdata.alc_tx_ring_paddr;
 	CSR_WRITE_4(sc, ALC_TX_BASE_ADDR_HI, ALC_ADDR_HI(paddr));
 	CSR_WRITE_4(sc, ALC_TDL_HEAD_ADDR_LO, ALC_ADDR_LO(paddr));
 	/* We don't use high priority ring. */
 	CSR_WRITE_4(sc, ALC_TDH_HEAD_ADDR_LO, 0);
 	/* Set Tx descriptor counter. */
 	CSR_WRITE_4(sc, ALC_TD_RING_CNT,
 	    (ALC_TX_RING_CNT << TD_RING_CNT_SHIFT) & TD_RING_CNT_MASK);
 	/* Set Rx descriptor base addresses. */
 	paddr = sc->alc_rdata.alc_rx_ring_paddr;
 	CSR_WRITE_4(sc, ALC_RX_BASE_ADDR_HI, ALC_ADDR_HI(paddr));
 	CSR_WRITE_4(sc, ALC_RD0_HEAD_ADDR_LO, ALC_ADDR_LO(paddr));
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		/* We use one Rx ring. */
 		CSR_WRITE_4(sc, ALC_RD1_HEAD_ADDR_LO, 0);
 		CSR_WRITE_4(sc, ALC_RD2_HEAD_ADDR_LO, 0);
 		CSR_WRITE_4(sc, ALC_RD3_HEAD_ADDR_LO, 0);
 	}
 	/* Set Rx descriptor counter. */
 	CSR_WRITE_4(sc, ALC_RD_RING_CNT,
 	    (ALC_RX_RING_CNT << RD_RING_CNT_SHIFT) & RD_RING_CNT_MASK);
 
 	/*
 	 * Let hardware split jumbo frames into alc_max_buf_sized chunks.
 	 * if it do not fit the buffer size. Rx return descriptor holds
 	 * a counter that indicates how many fragments were made by the
 	 * hardware. The buffer size should be multiple of 8 bytes.
 	 * Since hardware has limit on the size of buffer size, always
 	 * use the maximum value.
 	 * For strict-alignment architectures make sure to reduce buffer
 	 * size by 8 bytes to make room for alignment fixup.
 	 */
 #ifndef __NO_STRICT_ALIGNMENT
 	sc->alc_buf_size = RX_BUF_SIZE_MAX - sizeof(uint64_t);
 #else
 	sc->alc_buf_size = RX_BUF_SIZE_MAX;
 #endif
 	CSR_WRITE_4(sc, ALC_RX_BUF_SIZE, sc->alc_buf_size);
 
 	paddr = sc->alc_rdata.alc_rr_ring_paddr;
 	/* Set Rx return descriptor base addresses. */
 	CSR_WRITE_4(sc, ALC_RRD0_HEAD_ADDR_LO, ALC_ADDR_LO(paddr));
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		/* We use one Rx return ring. */
 		CSR_WRITE_4(sc, ALC_RRD1_HEAD_ADDR_LO, 0);
 		CSR_WRITE_4(sc, ALC_RRD2_HEAD_ADDR_LO, 0);
 		CSR_WRITE_4(sc, ALC_RRD3_HEAD_ADDR_LO, 0);
 	}
 	/* Set Rx return descriptor counter. */
 	CSR_WRITE_4(sc, ALC_RRD_RING_CNT,
 	    (ALC_RR_RING_CNT << RRD_RING_CNT_SHIFT) & RRD_RING_CNT_MASK);
 	paddr = sc->alc_rdata.alc_cmb_paddr;
 	CSR_WRITE_4(sc, ALC_CMB_BASE_ADDR_LO, ALC_ADDR_LO(paddr));
 	paddr = sc->alc_rdata.alc_smb_paddr;
 	CSR_WRITE_4(sc, ALC_SMB_BASE_ADDR_HI, ALC_ADDR_HI(paddr));
 	CSR_WRITE_4(sc, ALC_SMB_BASE_ADDR_LO, ALC_ADDR_LO(paddr));
 
 	if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B) {
 		/* Reconfigure SRAM - Vendor magic. */
 		CSR_WRITE_4(sc, ALC_SRAM_RX_FIFO_LEN, 0x000002A0);
 		CSR_WRITE_4(sc, ALC_SRAM_TX_FIFO_LEN, 0x00000100);
 		CSR_WRITE_4(sc, ALC_SRAM_RX_FIFO_ADDR, 0x029F0000);
 		CSR_WRITE_4(sc, ALC_SRAM_RD0_ADDR, 0x02BF02A0);
 		CSR_WRITE_4(sc, ALC_SRAM_TX_FIFO_ADDR, 0x03BF02C0);
 		CSR_WRITE_4(sc, ALC_SRAM_TD_ADDR, 0x03DF03C0);
 		CSR_WRITE_4(sc, ALC_TXF_WATER_MARK, 0x00000000);
 		CSR_WRITE_4(sc, ALC_RD_DMA_CFG, 0x00000000);
 	}
 
 	/* Tell hardware that we're ready to load DMA blocks. */
 	CSR_WRITE_4(sc, ALC_DMA_BLOCK, DMA_BLOCK_LOAD);
 
 	/* Configure interrupt moderation timer. */
 	reg = ALC_USECS(sc->alc_int_rx_mod) << IM_TIMER_RX_SHIFT;
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0)
 		reg |= ALC_USECS(sc->alc_int_tx_mod) << IM_TIMER_TX_SHIFT;
 	CSR_WRITE_4(sc, ALC_IM_TIMER, reg);
 	/*
 	 * We don't want to automatic interrupt clear as task queue
 	 * for the interrupt should know interrupt status.
 	 */
 	reg = CSR_READ_4(sc, ALC_MASTER_CFG);
 	reg &= ~(MASTER_IM_RX_TIMER_ENB | MASTER_IM_TX_TIMER_ENB);
 	reg |= MASTER_SA_TIMER_ENB;
 	if (ALC_USECS(sc->alc_int_rx_mod) != 0)
 		reg |= MASTER_IM_RX_TIMER_ENB;
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0 &&
 	    ALC_USECS(sc->alc_int_tx_mod) != 0)
 		reg |= MASTER_IM_TX_TIMER_ENB;
 	CSR_WRITE_4(sc, ALC_MASTER_CFG, reg);
 	/*
 	 * Disable interrupt re-trigger timer. We don't want automatic
 	 * re-triggering of un-ACKed interrupts.
 	 */
 	CSR_WRITE_4(sc, ALC_INTR_RETRIG_TIMER, ALC_USECS(0));
 	/* Configure CMB. */
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		CSR_WRITE_4(sc, ALC_CMB_TD_THRESH, ALC_TX_RING_CNT / 3);
 		CSR_WRITE_4(sc, ALC_CMB_TX_TIMER,
 		    ALC_USECS(sc->alc_int_tx_mod));
 	} else {
 		if ((sc->alc_flags & ALC_FLAG_CMB_BUG) == 0) {
 			CSR_WRITE_4(sc, ALC_CMB_TD_THRESH, 4);
 			CSR_WRITE_4(sc, ALC_CMB_TX_TIMER, ALC_USECS(5000));
 		} else
 			CSR_WRITE_4(sc, ALC_CMB_TX_TIMER, ALC_USECS(0));
 	}
 	/*
 	 * Hardware can be configured to issue SMB interrupt based
 	 * on programmed interval. Since there is a callout that is
 	 * invoked for every hz in driver we use that instead of
 	 * relying on periodic SMB interrupt.
 	 */
 	CSR_WRITE_4(sc, ALC_SMB_STAT_TIMER, ALC_USECS(0));
 	/* Clear MAC statistics. */
 	alc_stats_clear(sc);
 
 	/*
 	 * Always use maximum frame size that controller can support.
 	 * Otherwise received frames that has larger frame length
 	 * than alc(4) MTU would be silently dropped in hardware. This
 	 * would make path-MTU discovery hard as sender wouldn't get
 	 * any responses from receiver. alc(4) supports
 	 * multi-fragmented frames on Rx path so it has no issue on
 	 * assembling fragmented frames. Using maximum frame size also
 	 * removes the need to reinitialize hardware when interface
 	 * MTU configuration was changed.
 	 *
 	 * Be conservative in what you do, be liberal in what you
 	 * accept from others - RFC 793.
 	 */
 	CSR_WRITE_4(sc, ALC_FRAME_SIZE, sc->alc_ident->max_framelen);
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		/* Disable header split(?) */
 		CSR_WRITE_4(sc, ALC_HDS_CFG, 0);
 
 		/* Configure IPG/IFG parameters. */
 		CSR_WRITE_4(sc, ALC_IPG_IFG_CFG,
 		    ((IPG_IFG_IPGT_DEFAULT << IPG_IFG_IPGT_SHIFT) &
 		    IPG_IFG_IPGT_MASK) |
 		    ((IPG_IFG_MIFG_DEFAULT << IPG_IFG_MIFG_SHIFT) &
 		    IPG_IFG_MIFG_MASK) |
 		    ((IPG_IFG_IPG1_DEFAULT << IPG_IFG_IPG1_SHIFT) &
 		    IPG_IFG_IPG1_MASK) |
 		    ((IPG_IFG_IPG2_DEFAULT << IPG_IFG_IPG2_SHIFT) &
 		    IPG_IFG_IPG2_MASK));
 		/* Set parameters for half-duplex media. */
 		CSR_WRITE_4(sc, ALC_HDPX_CFG,
 		    ((HDPX_CFG_LCOL_DEFAULT << HDPX_CFG_LCOL_SHIFT) &
 		    HDPX_CFG_LCOL_MASK) |
 		    ((HDPX_CFG_RETRY_DEFAULT << HDPX_CFG_RETRY_SHIFT) &
 		    HDPX_CFG_RETRY_MASK) | HDPX_CFG_EXC_DEF_EN |
 		    ((HDPX_CFG_ABEBT_DEFAULT << HDPX_CFG_ABEBT_SHIFT) &
 		    HDPX_CFG_ABEBT_MASK) |
 		    ((HDPX_CFG_JAMIPG_DEFAULT << HDPX_CFG_JAMIPG_SHIFT) &
 		    HDPX_CFG_JAMIPG_MASK));
 	}
 
 	/*
 	 * Set TSO/checksum offload threshold. For frames that is
 	 * larger than this threshold, hardware wouldn't do
 	 * TSO/checksum offloading.
 	 */
 	reg = (sc->alc_ident->max_framelen >> TSO_OFFLOAD_THRESH_UNIT_SHIFT) &
 	    TSO_OFFLOAD_THRESH_MASK;
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		reg |= TSO_OFFLOAD_ERRLGPKT_DROP_ENB;
 	CSR_WRITE_4(sc, ALC_TSO_OFFLOAD_THRESH, reg);
 	/* Configure TxQ. */
 	reg = (alc_dma_burst[sc->alc_dma_rd_burst] <<
 	    TXQ_CFG_TX_FIFO_BURST_SHIFT) & TXQ_CFG_TX_FIFO_BURST_MASK;
 	if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B2)
 		reg >>= 1;
 	reg |= (TXQ_CFG_TD_BURST_DEFAULT << TXQ_CFG_TD_BURST_SHIFT) &
 	    TXQ_CFG_TD_BURST_MASK;
 	reg |= TXQ_CFG_IP_OPTION_ENB | TXQ_CFG_8023_ENB;
 	CSR_WRITE_4(sc, ALC_TXQ_CFG, reg | TXQ_CFG_ENHANCED_MODE);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		reg = (TXQ_CFG_TD_BURST_DEFAULT << HQTD_CFG_Q1_BURST_SHIFT |
 		    TXQ_CFG_TD_BURST_DEFAULT << HQTD_CFG_Q2_BURST_SHIFT |
 		    TXQ_CFG_TD_BURST_DEFAULT << HQTD_CFG_Q3_BURST_SHIFT |
 		    HQTD_CFG_BURST_ENB);
 		CSR_WRITE_4(sc, ALC_HQTD_CFG, reg);
 		reg = WRR_PRI_RESTRICT_NONE;
 		reg |= (WRR_PRI_DEFAULT << WRR_PRI0_SHIFT |
 		    WRR_PRI_DEFAULT << WRR_PRI1_SHIFT |
 		    WRR_PRI_DEFAULT << WRR_PRI2_SHIFT |
 		    WRR_PRI_DEFAULT << WRR_PRI3_SHIFT);
 		CSR_WRITE_4(sc, ALC_WRR, reg);
 	} else {
 		/* Configure Rx free descriptor pre-fetching. */
 		CSR_WRITE_4(sc, ALC_RX_RD_FREE_THRESH,
 		    ((RX_RD_FREE_THRESH_HI_DEFAULT <<
 		    RX_RD_FREE_THRESH_HI_SHIFT) & RX_RD_FREE_THRESH_HI_MASK) |
 		    ((RX_RD_FREE_THRESH_LO_DEFAULT <<
 		    RX_RD_FREE_THRESH_LO_SHIFT) & RX_RD_FREE_THRESH_LO_MASK));
 	}
 
 	/*
 	 * Configure flow control parameters.
 	 * XON  : 80% of Rx FIFO
 	 * XOFF : 30% of Rx FIFO
 	 */
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		reg = CSR_READ_4(sc, ALC_SRAM_RX_FIFO_LEN);
 		reg &= SRAM_RX_FIFO_LEN_MASK;
 		reg *= 8;
 		if (reg > 8 * 1024)
 			reg -= RX_FIFO_PAUSE_816X_RSVD;
 		else
 			reg -= RX_BUF_SIZE_MAX;
 		reg /= 8;
 		CSR_WRITE_4(sc, ALC_RX_FIFO_PAUSE_THRESH,
 		    ((reg << RX_FIFO_PAUSE_THRESH_LO_SHIFT) &
 		    RX_FIFO_PAUSE_THRESH_LO_MASK) |
 		    (((RX_FIFO_PAUSE_816X_RSVD / 8) <<
 		    RX_FIFO_PAUSE_THRESH_HI_SHIFT) &
 		    RX_FIFO_PAUSE_THRESH_HI_MASK));
 	} else if (sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8131 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8132) {
 		reg = CSR_READ_4(sc, ALC_SRAM_RX_FIFO_LEN);
 		rxf_hi = (reg * 8) / 10;
 		rxf_lo = (reg * 3) / 10;
 		CSR_WRITE_4(sc, ALC_RX_FIFO_PAUSE_THRESH,
 		    ((rxf_lo << RX_FIFO_PAUSE_THRESH_LO_SHIFT) &
 		     RX_FIFO_PAUSE_THRESH_LO_MASK) |
 		    ((rxf_hi << RX_FIFO_PAUSE_THRESH_HI_SHIFT) &
 		     RX_FIFO_PAUSE_THRESH_HI_MASK));
 	}
 
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		/* Disable RSS until I understand L1C/L2C's RSS logic. */
 		CSR_WRITE_4(sc, ALC_RSS_IDT_TABLE0, 0);
 		CSR_WRITE_4(sc, ALC_RSS_CPU, 0);
 	}
 
 	/* Configure RxQ. */
 	reg = (RXQ_CFG_RD_BURST_DEFAULT << RXQ_CFG_RD_BURST_SHIFT) &
 	    RXQ_CFG_RD_BURST_MASK;
 	reg |= RXQ_CFG_RSS_MODE_DIS;
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0)
 		reg |= (RXQ_CFG_816X_IDT_TBL_SIZE_DEFAULT <<
 		    RXQ_CFG_816X_IDT_TBL_SIZE_SHIFT) &
 		    RXQ_CFG_816X_IDT_TBL_SIZE_MASK;
 	if ((sc->alc_flags & ALC_FLAG_FASTETHER) == 0 &&
 	    sc->alc_ident->deviceid != DEVICEID_ATHEROS_AR8151_V2)
 		reg |= RXQ_CFG_ASPM_THROUGHPUT_LIMIT_1M;
 	CSR_WRITE_4(sc, ALC_RXQ_CFG, reg);
 
 	/* Configure DMA parameters. */
 	reg = DMA_CFG_OUT_ORDER | DMA_CFG_RD_REQ_PRI;
 	reg |= sc->alc_rcb;
 	if ((sc->alc_flags & ALC_FLAG_CMB_BUG) == 0)
 		reg |= DMA_CFG_CMB_ENB;
 	if ((sc->alc_flags & ALC_FLAG_SMB_BUG) == 0)
 		reg |= DMA_CFG_SMB_ENB;
 	else
 		reg |= DMA_CFG_SMB_DIS;
 	reg |= (sc->alc_dma_rd_burst & DMA_CFG_RD_BURST_MASK) <<
 	    DMA_CFG_RD_BURST_SHIFT;
 	reg |= (sc->alc_dma_wr_burst & DMA_CFG_WR_BURST_MASK) <<
 	    DMA_CFG_WR_BURST_SHIFT;
 	reg |= (DMA_CFG_RD_DELAY_CNT_DEFAULT << DMA_CFG_RD_DELAY_CNT_SHIFT) &
 	    DMA_CFG_RD_DELAY_CNT_MASK;
 	reg |= (DMA_CFG_WR_DELAY_CNT_DEFAULT << DMA_CFG_WR_DELAY_CNT_SHIFT) &
 	    DMA_CFG_WR_DELAY_CNT_MASK;
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0) {
 		switch (AR816X_REV(sc->alc_rev)) {
 		case AR816X_REV_A0:
 		case AR816X_REV_A1:
 			reg |= DMA_CFG_RD_CHNL_SEL_1;
 			break;
 		case AR816X_REV_B0:
 			/* FALLTHROUGH */
 		default:
 			reg |= DMA_CFG_RD_CHNL_SEL_3;
 			break;
 		}
 	}
 	CSR_WRITE_4(sc, ALC_DMA_CFG, reg);
 
 	/*
 	 * Configure Tx/Rx MACs.
 	 *  - Auto-padding for short frames.
 	 *  - Enable CRC generation.
 	 *  Actual reconfiguration of MAC for resolved speed/duplex
 	 *  is followed after detection of link establishment.
 	 *  AR813x/AR815x always does checksum computation regardless
 	 *  of MAC_CFG_RXCSUM_ENB bit. Also the controller is known to
 	 *  have bug in protocol field in Rx return structure so
 	 *  these controllers can't handle fragmented frames. Disable
 	 *  Rx checksum offloading until there is a newer controller
 	 *  that has sane implementation.
 	 */
 	reg = MAC_CFG_TX_CRC_ENB | MAC_CFG_TX_AUTO_PAD | MAC_CFG_FULL_DUPLEX |
 	    ((MAC_CFG_PREAMBLE_DEFAULT << MAC_CFG_PREAMBLE_SHIFT) &
 	    MAC_CFG_PREAMBLE_MASK);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) != 0 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8151_V2 ||
 	    sc->alc_ident->deviceid == DEVICEID_ATHEROS_AR8152_B2)
 		reg |= MAC_CFG_HASH_ALG_CRC32 | MAC_CFG_SPEED_MODE_SW;
 	if ((sc->alc_flags & ALC_FLAG_FASTETHER) != 0)
 		reg |= MAC_CFG_SPEED_10_100;
 	else
 		reg |= MAC_CFG_SPEED_1000;
 	CSR_WRITE_4(sc, ALC_MAC_CFG, reg);
 
 	/* Set up the receive filter. */
 	alc_rxfilter(sc);
 	alc_rxvlan(sc);
 
 	/* Acknowledge all pending interrupts and clear it. */
 	CSR_WRITE_4(sc, ALC_INTR_MASK, ALC_INTRS);
 	CSR_WRITE_4(sc, ALC_INTR_STATUS, 0xFFFFFFFF);
 	CSR_WRITE_4(sc, ALC_INTR_STATUS, 0);
 
 	ifp->if_drv_flags |= IFF_DRV_RUNNING;
 	ifp->if_drv_flags &= ~IFF_DRV_OACTIVE;
 
 	sc->alc_flags &= ~ALC_FLAG_LINK;
 	/* Switch to the current media. */
 	alc_mediachange_locked(sc);
 
 	callout_reset(&sc->alc_tick_ch, hz, alc_tick, sc);
 }
 
 static void
 alc_stop(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	struct alc_txdesc *txd;
 	struct alc_rxdesc *rxd;
 	uint32_t reg;
 	int i;
 
 	ALC_LOCK_ASSERT(sc);
 	/*
 	 * Mark the interface down and cancel the watchdog timer.
 	 */
 	ifp = sc->alc_ifp;
 	ifp->if_drv_flags &= ~(IFF_DRV_RUNNING | IFF_DRV_OACTIVE);
 	sc->alc_flags &= ~ALC_FLAG_LINK;
 	callout_stop(&sc->alc_tick_ch);
 	sc->alc_watchdog_timer = 0;
 	alc_stats_update(sc);
 	/* Disable interrupts. */
 	CSR_WRITE_4(sc, ALC_INTR_MASK, 0);
 	CSR_WRITE_4(sc, ALC_INTR_STATUS, 0xFFFFFFFF);
 	/* Disable DMA. */
 	reg = CSR_READ_4(sc, ALC_DMA_CFG);
 	reg &= ~(DMA_CFG_CMB_ENB | DMA_CFG_SMB_ENB);
 	reg |= DMA_CFG_SMB_DIS;
 	CSR_WRITE_4(sc, ALC_DMA_CFG, reg);
 	DELAY(1000);
 	/* Stop Rx/Tx MACs. */
 	alc_stop_mac(sc);
 	/* Disable interrupts which might be touched in taskq handler. */
 	CSR_WRITE_4(sc, ALC_INTR_STATUS, 0xFFFFFFFF);
 	/* Disable L0s/L1s */
 	alc_aspm(sc, 0, IFM_UNKNOWN);
 	/* Reclaim Rx buffers that have been processed. */
 	if (sc->alc_cdata.alc_rxhead != NULL)
 		m_freem(sc->alc_cdata.alc_rxhead);
 	ALC_RXCHAIN_RESET(sc);
 	/*
 	 * Free Tx/Rx mbufs still in the queues.
 	 */
 	for (i = 0; i < ALC_RX_RING_CNT; i++) {
 		rxd = &sc->alc_cdata.alc_rxdesc[i];
 		if (rxd->rx_m != NULL) {
 			bus_dmamap_sync(sc->alc_cdata.alc_rx_tag,
 			    rxd->rx_dmamap, BUS_DMASYNC_POSTREAD);
 			bus_dmamap_unload(sc->alc_cdata.alc_rx_tag,
 			    rxd->rx_dmamap);
 			m_freem(rxd->rx_m);
 			rxd->rx_m = NULL;
 		}
 	}
 	for (i = 0; i < ALC_TX_RING_CNT; i++) {
 		txd = &sc->alc_cdata.alc_txdesc[i];
 		if (txd->tx_m != NULL) {
 			bus_dmamap_sync(sc->alc_cdata.alc_tx_tag,
 			    txd->tx_dmamap, BUS_DMASYNC_POSTWRITE);
 			bus_dmamap_unload(sc->alc_cdata.alc_tx_tag,
 			    txd->tx_dmamap);
 			m_freem(txd->tx_m);
 			txd->tx_m = NULL;
 		}
 	}
 }
 
 static void
 alc_stop_mac(struct alc_softc *sc)
 {
 	uint32_t reg;
 	int i;
 
 	alc_stop_queue(sc);
 	/* Disable Rx/Tx MAC. */
 	reg = CSR_READ_4(sc, ALC_MAC_CFG);
 	if ((reg & (MAC_CFG_TX_ENB | MAC_CFG_RX_ENB)) != 0) {
 		reg &= ~(MAC_CFG_TX_ENB | MAC_CFG_RX_ENB);
 		CSR_WRITE_4(sc, ALC_MAC_CFG, reg);
 	}
 	for (i = ALC_TIMEOUT; i > 0; i--) {
 		reg = CSR_READ_4(sc, ALC_IDLE_STATUS);
 		if ((reg & (IDLE_STATUS_RXMAC | IDLE_STATUS_TXMAC)) == 0)
 			break;
 		DELAY(10);
 	}
 	if (i == 0)
 		device_printf(sc->alc_dev,
 		    "could not disable Rx/Tx MAC(0x%08x)!\n", reg);
 }
 
 static void
 alc_start_queue(struct alc_softc *sc)
 {
 	uint32_t qcfg[] = {
 		0,
 		RXQ_CFG_QUEUE0_ENB,
 		RXQ_CFG_QUEUE0_ENB | RXQ_CFG_QUEUE1_ENB,
 		RXQ_CFG_QUEUE0_ENB | RXQ_CFG_QUEUE1_ENB | RXQ_CFG_QUEUE2_ENB,
 		RXQ_CFG_ENB
 	};
 	uint32_t cfg;
 
 	ALC_LOCK_ASSERT(sc);
 
 	/* Enable RxQ. */
 	cfg = CSR_READ_4(sc, ALC_RXQ_CFG);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		cfg &= ~RXQ_CFG_ENB;
 		cfg |= qcfg[1];
 	} else
 		cfg |= RXQ_CFG_QUEUE0_ENB;
 	CSR_WRITE_4(sc, ALC_RXQ_CFG, cfg);
 	/* Enable TxQ. */
 	cfg = CSR_READ_4(sc, ALC_TXQ_CFG);
 	cfg |= TXQ_CFG_ENB;
 	CSR_WRITE_4(sc, ALC_TXQ_CFG, cfg);
 }
 
 static void
 alc_stop_queue(struct alc_softc *sc)
 {
 	uint32_t reg;
 	int i;
 
 	/* Disable RxQ. */
 	reg = CSR_READ_4(sc, ALC_RXQ_CFG);
 	if ((sc->alc_flags & ALC_FLAG_AR816X_FAMILY) == 0) {
 		if ((reg & RXQ_CFG_ENB) != 0) {
 			reg &= ~RXQ_CFG_ENB;
 			CSR_WRITE_4(sc, ALC_RXQ_CFG, reg);
 		}
 	} else {
 		if ((reg & RXQ_CFG_QUEUE0_ENB) != 0) {
 			reg &= ~RXQ_CFG_QUEUE0_ENB;
 			CSR_WRITE_4(sc, ALC_RXQ_CFG, reg);
 		}
 	}
 	/* Disable TxQ. */
 	reg = CSR_READ_4(sc, ALC_TXQ_CFG);
 	if ((reg & TXQ_CFG_ENB) != 0) {
 		reg &= ~TXQ_CFG_ENB;
 		CSR_WRITE_4(sc, ALC_TXQ_CFG, reg);
 	}
 	DELAY(40);
 	for (i = ALC_TIMEOUT; i > 0; i--) {
 		reg = CSR_READ_4(sc, ALC_IDLE_STATUS);
 		if ((reg & (IDLE_STATUS_RXQ | IDLE_STATUS_TXQ)) == 0)
 			break;
 		DELAY(10);
 	}
 	if (i == 0)
 		device_printf(sc->alc_dev,
 		    "could not disable RxQ/TxQ (0x%08x)!\n", reg);
 }
 
 static void
 alc_init_tx_ring(struct alc_softc *sc)
 {
 	struct alc_ring_data *rd;
 	struct alc_txdesc *txd;
 	int i;
 
 	ALC_LOCK_ASSERT(sc);
 
 	sc->alc_cdata.alc_tx_prod = 0;
 	sc->alc_cdata.alc_tx_cons = 0;
 	sc->alc_cdata.alc_tx_cnt = 0;
 
 	rd = &sc->alc_rdata;
 	bzero(rd->alc_tx_ring, ALC_TX_RING_SZ);
 	for (i = 0; i < ALC_TX_RING_CNT; i++) {
 		txd = &sc->alc_cdata.alc_txdesc[i];
 		txd->tx_m = NULL;
 	}
 
 	bus_dmamap_sync(sc->alc_cdata.alc_tx_ring_tag,
 	    sc->alc_cdata.alc_tx_ring_map, BUS_DMASYNC_PREWRITE);
 }
 
 static int
 alc_init_rx_ring(struct alc_softc *sc)
 {
 	struct alc_ring_data *rd;
 	struct alc_rxdesc *rxd;
 	int i;
 
 	ALC_LOCK_ASSERT(sc);
 
 	sc->alc_cdata.alc_rx_cons = ALC_RX_RING_CNT - 1;
 	sc->alc_morework = 0;
 	rd = &sc->alc_rdata;
 	bzero(rd->alc_rx_ring, ALC_RX_RING_SZ);
 	for (i = 0; i < ALC_RX_RING_CNT; i++) {
 		rxd = &sc->alc_cdata.alc_rxdesc[i];
 		rxd->rx_m = NULL;
 		rxd->rx_desc = &rd->alc_rx_ring[i];
 		if (alc_newbuf(sc, rxd) != 0)
 			return (ENOBUFS);
 	}
 
 	/*
 	 * Since controller does not update Rx descriptors, driver
 	 * does have to read Rx descriptors back so BUS_DMASYNC_PREWRITE
 	 * is enough to ensure coherence.
 	 */
 	bus_dmamap_sync(sc->alc_cdata.alc_rx_ring_tag,
 	    sc->alc_cdata.alc_rx_ring_map, BUS_DMASYNC_PREWRITE);
 	/* Let controller know availability of new Rx buffers. */
 	CSR_WRITE_4(sc, ALC_MBOX_RD0_PROD_IDX, sc->alc_cdata.alc_rx_cons);
 
 	return (0);
 }
 
 static void
 alc_init_rr_ring(struct alc_softc *sc)
 {
 	struct alc_ring_data *rd;
 
 	ALC_LOCK_ASSERT(sc);
 
 	sc->alc_cdata.alc_rr_cons = 0;
 	ALC_RXCHAIN_RESET(sc);
 
 	rd = &sc->alc_rdata;
 	bzero(rd->alc_rr_ring, ALC_RR_RING_SZ);
 	bus_dmamap_sync(sc->alc_cdata.alc_rr_ring_tag,
 	    sc->alc_cdata.alc_rr_ring_map,
 	    BUS_DMASYNC_PREREAD | BUS_DMASYNC_PREWRITE);
 }
 
 static void
 alc_init_cmb(struct alc_softc *sc)
 {
 	struct alc_ring_data *rd;
 
 	ALC_LOCK_ASSERT(sc);
 
 	rd = &sc->alc_rdata;
 	bzero(rd->alc_cmb, ALC_CMB_SZ);
 	bus_dmamap_sync(sc->alc_cdata.alc_cmb_tag, sc->alc_cdata.alc_cmb_map,
 	    BUS_DMASYNC_PREREAD | BUS_DMASYNC_PREWRITE);
 }
 
 static void
 alc_init_smb(struct alc_softc *sc)
 {
 	struct alc_ring_data *rd;
 
 	ALC_LOCK_ASSERT(sc);
 
 	rd = &sc->alc_rdata;
 	bzero(rd->alc_smb, ALC_SMB_SZ);
 	bus_dmamap_sync(sc->alc_cdata.alc_smb_tag, sc->alc_cdata.alc_smb_map,
 	    BUS_DMASYNC_PREREAD | BUS_DMASYNC_PREWRITE);
 }
 
 static void
 alc_rxvlan(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	uint32_t reg;
 
 	ALC_LOCK_ASSERT(sc);
 
 	ifp = sc->alc_ifp;
 	reg = CSR_READ_4(sc, ALC_MAC_CFG);
 	if ((ifp->if_capenable & IFCAP_VLAN_HWTAGGING) != 0)
 		reg |= MAC_CFG_VLAN_TAG_STRIP;
 	else
 		reg &= ~MAC_CFG_VLAN_TAG_STRIP;
 	CSR_WRITE_4(sc, ALC_MAC_CFG, reg);
 }
 
 static void
 alc_rxfilter(struct alc_softc *sc)
 {
 	struct ifnet *ifp;
 	struct ifmultiaddr *ifma;
 	uint32_t crc;
 	uint32_t mchash[2];
 	uint32_t rxcfg;
 
 	ALC_LOCK_ASSERT(sc);
 
 	ifp = sc->alc_ifp;
 
 	bzero(mchash, sizeof(mchash));
 	rxcfg = CSR_READ_4(sc, ALC_MAC_CFG);
 	rxcfg &= ~(MAC_CFG_ALLMULTI | MAC_CFG_BCAST | MAC_CFG_PROMISC);
 	if ((ifp->if_flags & IFF_BROADCAST) != 0)
 		rxcfg |= MAC_CFG_BCAST;
 	if ((ifp->if_flags & (IFF_PROMISC | IFF_ALLMULTI)) != 0) {
 		if ((ifp->if_flags & IFF_PROMISC) != 0)
 			rxcfg |= MAC_CFG_PROMISC;
 		if ((ifp->if_flags & IFF_ALLMULTI) != 0)
 			rxcfg |= MAC_CFG_ALLMULTI;
 		mchash[0] = 0xFFFFFFFF;
 		mchash[1] = 0xFFFFFFFF;
 		goto chipit;
 	}
 
 	if_maddr_rlock(ifp);
 	TAILQ_FOREACH(ifma, &sc->alc_ifp->if_multiaddrs, ifma_link) {
 		if (ifma->ifma_addr->sa_family != AF_LINK)
 			continue;
 		crc = ether_crc32_be(LLADDR((struct sockaddr_dl *)
 		    ifma->ifma_addr), ETHER_ADDR_LEN);
 		mchash[crc >> 31] |= 1 << ((crc >> 26) & 0x1f);
 	}
 	if_maddr_runlock(ifp);
 
 chipit:
 	CSR_WRITE_4(sc, ALC_MAR0, mchash[0]);
 	CSR_WRITE_4(sc, ALC_MAR1, mchash[1]);
 	CSR_WRITE_4(sc, ALC_MAC_CFG, rxcfg);
 }
 
 static int
 sysctl_int_range(SYSCTL_HANDLER_ARGS, int low, int high)
 {
 	int error, value;
 
 	if (arg1 == NULL)
 		return (EINVAL);
 	value = *(int *)arg1;
 	error = sysctl_handle_int(oidp, &value, 0, req);
 	if (error || req->newptr == NULL)
 		return (error);
 	if (value < low || value > high)
 		return (EINVAL);
 	*(int *)arg1 = value;
 
 	return (0);
 }
 
 static int
 sysctl_hw_alc_proc_limit(SYSCTL_HANDLER_ARGS)
 {
 	return (sysctl_int_range(oidp, arg1, arg2, req,
 	    ALC_PROC_MIN, ALC_PROC_MAX));
 }
 
 static int
 sysctl_hw_alc_int_mod(SYSCTL_HANDLER_ARGS)
 {
 
 	return (sysctl_int_range(oidp, arg1, arg2, req,
 	    ALC_IM_TIMER_MIN, ALC_IM_TIMER_MAX));
 }
Index: projects/clang360-import/sys/dev/ofw/openfirm.c
===================================================================
--- projects/clang360-import/sys/dev/ofw/openfirm.c	(revision 277944)
+++ projects/clang360-import/sys/dev/ofw/openfirm.c	(revision 277945)
@@ -1,791 +1,796 @@
 /*	$NetBSD: Locore.c,v 1.7 2000/08/20 07:04:59 tsubai Exp $	*/
 
 /*-
  * Copyright (C) 1995, 1996 Wolfgang Solfrank.
  * Copyright (C) 1995, 1996 TooLs GmbH.
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  * 3. All advertising materials mentioning features or use of this software
  *    must display the following acknowledgement:
  *	This product includes software developed by TooLs GmbH.
  * 4. The name of TooLs GmbH may not be used to endorse or promote products
  *    derived from this software without specific prior written permission.
  *
  * THIS SOFTWARE IS PROVIDED BY TOOLS GMBH ``AS IS'' AND ANY EXPRESS OR
  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
  * IN NO EVENT SHALL TOOLS GMBH BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
  * SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
  * PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
  * OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY,
  * WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR
  * OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF
  * ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  */
 /*-
  * Copyright (C) 2000 Benno Rice.
  * All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY Benno Rice ``AS IS'' AND ANY EXPRESS OR
  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
  * IN NO EVENT SHALL TOOLS GMBH BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
  * SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
  * PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS;
  * OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY,
  * WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR
  * OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF
  * ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include "opt_platform.h"
 
 #include <sys/param.h>
 #include <sys/kernel.h>
 #include <sys/lock.h>
 #include <sys/malloc.h>
 #include <sys/mutex.h>
 #include <sys/queue.h>
 #include <sys/systm.h>
 #include <sys/endian.h>
 
 #include <machine/stdarg.h>
 
 #include <dev/ofw/ofwvar.h>
 #include <dev/ofw/openfirm.h>
 
 #include "ofw_if.h"
 
 static void OF_putchar(int c, void *arg);
 
 MALLOC_DEFINE(M_OFWPROP, "openfirm", "Open Firmware properties");
 
 static ihandle_t stdout;
 
 static ofw_def_t	*ofw_def_impl = NULL;
 static ofw_t		ofw_obj;
 static struct ofw_kobj	ofw_kernel_obj;
 static struct kobj_ops	ofw_kernel_kops;
 
 struct xrefinfo {
 	phandle_t	xref;
 	phandle_t 	node;
 	device_t  	dev;
 	SLIST_ENTRY(xrefinfo) next_entry;
 };
 
 static SLIST_HEAD(, xrefinfo) xreflist = SLIST_HEAD_INITIALIZER(xreflist);
 static struct mtx xreflist_lock;
 static boolean_t xref_init_done;
 
 #define	FIND_BY_XREF	0
 #define	FIND_BY_NODE	1
 #define	FIND_BY_DEV	2
 
 /*
  * xref-phandle-device lookup helper routines.
  *
  * As soon as we are able to use malloc(), walk the node tree and build a list
  * of info that cross-references node handles, xref handles, and device_t
  * instances.  This list exists primarily to allow association of a device_t
  * with an xref handle, but it is also used to speed up translation between xref
  * and node handles.  Before malloc() is available we have to recursively search
  * the node tree each time we want to translate between a node and xref handle.
  * Afterwards we can do the translations by searching this much shorter list.
  */
 static void
 xrefinfo_create(phandle_t node)
 {
 	struct xrefinfo * xi;
 	phandle_t child, xref;
 
 	/*
 	 * Recursively descend from parent, looking for nodes with a property
 	 * named either "phandle", "ibm,phandle", or "linux,phandle".  For each
 	 * such node found create an entry in the xreflist.
 	 */
 	for (child = OF_child(node); child != 0; child = OF_peer(child)) {
 		xrefinfo_create(child);
 		if (OF_getencprop(child, "phandle", &xref, sizeof(xref)) ==
 		    -1 && OF_getencprop(child, "ibm,phandle", &xref,
 		    sizeof(xref)) == -1 && OF_getencprop(child,
 		    "linux,phandle", &xref, sizeof(xref)) == -1)
 			continue;
 		xi = malloc(sizeof(*xi), M_OFWPROP, M_WAITOK | M_ZERO);
 		xi->node = child;
 		xi->xref = xref;
 		SLIST_INSERT_HEAD(&xreflist, xi, next_entry);
 	}
 }
 
 static void
 xrefinfo_init(void *unsed)
 {
 
 	/*
 	 * There is no locking during this init because it runs much earlier
 	 * than any of the clients/consumers of the xref list data, but we do
 	 * initialize the mutex that will be used for access later.
 	 */
 	mtx_init(&xreflist_lock, "OF xreflist lock", NULL, MTX_DEF);
 	xrefinfo_create(OF_peer(0));
 	xref_init_done = true;
 }
 SYSINIT(xrefinfo, SI_SUB_KMEM, SI_ORDER_ANY, xrefinfo_init, NULL);
 
 static struct xrefinfo *
 xrefinfo_find(phandle_t phandle, int find_by)
 {
 	struct xrefinfo *rv, *xi;
 
 	rv = NULL;
 	mtx_lock(&xreflist_lock);
 	SLIST_FOREACH(xi, &xreflist, next_entry) {
 		if ((find_by == FIND_BY_XREF && phandle == xi->xref) ||
 		    (find_by == FIND_BY_NODE && phandle == xi->node) ||
 		    (find_by == FIND_BY_DEV && phandle == (uintptr_t)xi->dev)) {
 			rv = xi;
 			break;
 		}
 	}
 	mtx_unlock(&xreflist_lock);
 	return (rv);
 }
 
 static struct xrefinfo *
 xrefinfo_add(phandle_t node, phandle_t xref, device_t dev)
 {
 	struct xrefinfo *xi;
 
 	xi = malloc(sizeof(*xi), M_OFWPROP, M_WAITOK);
 	xi->node = node;
 	xi->xref = xref;
 	xi->dev  = dev;
 	mtx_lock(&xreflist_lock);
 	SLIST_INSERT_HEAD(&xreflist, xi, next_entry);
 	mtx_unlock(&xreflist_lock);
 	return (xi);
 }
 
 /*
  * OFW install routines.  Highest priority wins, equal priority also
  * overrides allowing last-set to win.
  */
 SET_DECLARE(ofw_set, ofw_def_t);
 
 boolean_t
 OF_install(char *name, int prio)
 {
 	ofw_def_t *ofwp, **ofwpp;
 	static int curr_prio = 0;
 
 	/*
 	 * Try and locate the OFW kobj corresponding to the name.
 	 */
 	SET_FOREACH(ofwpp, ofw_set) {
 		ofwp = *ofwpp;
 
 		if (ofwp->name &&
 		    !strcmp(ofwp->name, name) &&
 		    prio >= curr_prio) {
 			curr_prio = prio;
 			ofw_def_impl = ofwp;
 			return (TRUE);
 		}
 	}
 
 	return (FALSE);
 }
 
 /* Initializer */
 int
 OF_init(void *cookie)
 {
 	phandle_t chosen;
 	int rv;
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	ofw_obj = &ofw_kernel_obj;
 	/*
 	 * Take care of compiling the selected class, and
 	 * then statically initialize the OFW object.
 	 */
 	kobj_class_compile_static(ofw_def_impl, &ofw_kernel_kops);
 	kobj_init_static((kobj_t)ofw_obj, ofw_def_impl);
 
 	rv = OFW_INIT(ofw_obj, cookie);
 
 	if ((chosen = OF_finddevice("/chosen")) != -1)
 		if (OF_getencprop(chosen, "stdout", &stdout,
 		    sizeof(stdout)) == -1)
 			stdout = -1;
 
 	return (rv);
 }
 
 static void
 OF_putchar(int c, void *arg __unused)
 {
 	char cbuf;
 
 	if (c == '\n') {
 		cbuf = '\r';
 		OF_write(stdout, &cbuf, 1);
 	}
 
 	cbuf = c;
 	OF_write(stdout, &cbuf, 1);
 }
 
 void
 OF_printf(const char *fmt, ...)
 {
 	va_list	va;
 
 	va_start(va, fmt);
 	(void)kvprintf(fmt, OF_putchar, NULL, 10, va);
 	va_end(va);
 }
 
 /*
  * Generic functions
  */
 
 /* Test to see if a service exists. */
 int
 OF_test(const char *name)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_TEST(ofw_obj, name));
 }
 
 int
 OF_interpret(const char *cmd, int nreturns, ...)
 {
 	va_list ap;
 	cell_t slots[16];
 	int i = 0;
 	int status;
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	status = OFW_INTERPRET(ofw_obj, cmd, nreturns, slots);
 	if (status == -1)
 		return (status);
 
 	va_start(ap, nreturns);
 	while (i < nreturns)
 		*va_arg(ap, cell_t *) = slots[i++];
 	va_end(ap);
 
 	return (status);
 }
 
 /*
  * Device tree functions
  */
 
 /* Return the next sibling of this node or 0. */
 phandle_t
 OF_peer(phandle_t node)
 {
 
 	if (ofw_def_impl == NULL)
 		return (0);
 
 	return (OFW_PEER(ofw_obj, node));
 }
 
 /* Return the first child of this node or 0. */
 phandle_t
 OF_child(phandle_t node)
 {
 
 	if (ofw_def_impl == NULL)
 		return (0);
 
 	return (OFW_CHILD(ofw_obj, node));
 }
 
 /* Return the parent of this node or 0. */
 phandle_t
 OF_parent(phandle_t node)
 {
 
 	if (ofw_def_impl == NULL)
 		return (0);
 
 	return (OFW_PARENT(ofw_obj, node));
 }
 
 /* Return the package handle that corresponds to an instance handle. */
 phandle_t
 OF_instance_to_package(ihandle_t instance)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_INSTANCE_TO_PACKAGE(ofw_obj, instance));
 }
 
 /* Get the length of a property of a package. */
 ssize_t
 OF_getproplen(phandle_t package, const char *propname)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_GETPROPLEN(ofw_obj, package, propname));
 }
 
 /* Check existence of a property of a package. */
 int
 OF_hasprop(phandle_t package, const char *propname)
 {
 
 	return (OF_getproplen(package, propname) >= 0 ? 1 : 0);
 }
 
 /* Get the value of a property of a package. */
 ssize_t
 OF_getprop(phandle_t package, const char *propname, void *buf, size_t buflen)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_GETPROP(ofw_obj, package, propname, buf, buflen));
 }
 
 ssize_t
 OF_getencprop(phandle_t node, const char *propname, pcell_t *buf, size_t len)
 {
 	ssize_t retval;
 	int i;
 
 	KASSERT(len % 4 == 0, ("Need a multiple of 4 bytes"));
 
 	retval = OF_getprop(node, propname, buf, len);
 	for (i = 0; i < len/4; i++)
 		buf[i] = be32toh(buf[i]);
 
 	return (retval);
 }
 
 /*
  * Recursively search the node and its parent for the given property, working
  * downward from the node to the device tree root.  Returns the value of the
  * first match.
  */
 ssize_t
 OF_searchprop(phandle_t node, const char *propname, void *buf, size_t len)
 {
 	ssize_t rv;
 
 	for (; node != 0; node = OF_parent(node))
 		if ((rv = OF_getprop(node, propname, buf, len)) != -1)
 			return (rv);
 	return (-1);
 }
 
 ssize_t
 OF_searchencprop(phandle_t node, const char *propname, void *buf, size_t len)
 {
 	ssize_t rv;
 
 	for (; node != 0; node = OF_parent(node))
 		if ((rv = OF_getencprop(node, propname, buf, len)) != -1)
 			return (rv);
 	return (-1);
 }
 
 /*
  * Store the value of a property of a package into newly allocated memory
  * (using the M_OFWPROP malloc pool and M_WAITOK).  elsz is the size of a
  * single element, the number of elements is return in number.
  */
 ssize_t
 OF_getprop_alloc(phandle_t package, const char *propname, int elsz, void **buf)
 {
 	int len;
 
 	*buf = NULL;
 	if ((len = OF_getproplen(package, propname)) == -1 ||
 	    len % elsz != 0)
 		return (-1);
 
 	*buf = malloc(len, M_OFWPROP, M_WAITOK);
 	if (OF_getprop(package, propname, *buf, len) == -1) {
 		free(*buf, M_OFWPROP);
 		*buf = NULL;
 		return (-1);
 	}
 	return (len / elsz);
 }
 
 ssize_t
 OF_getencprop_alloc(phandle_t package, const char *name, int elsz, void **buf)
 {
 	ssize_t retval;
 	pcell_t *cell;
 	int i;
 
 	retval = OF_getprop_alloc(package, name, elsz, buf);
-	if (retval == -1 || retval*elsz % 4 != 0)
+	if (retval == -1)
 		return (-1);
+ 	if (retval * elsz % 4 != 0) {
+		free(*buf, M_OFWPROP);
+		*buf = NULL;
+		return (-1);
+	}
 
 	cell = *buf;
-	for (i = 0; i < retval*elsz/4; i++)
+	for (i = 0; i < retval * elsz / 4; i++)
 		cell[i] = be32toh(cell[i]);
 
 	return (retval);
 }
 
 /* Get the next property of a package. */
 int
 OF_nextprop(phandle_t package, const char *previous, char *buf, size_t size)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_NEXTPROP(ofw_obj, package, previous, buf, size));
 }
 
 /* Set the value of a property of a package. */
 int
 OF_setprop(phandle_t package, const char *propname, const void *buf, size_t len)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_SETPROP(ofw_obj, package, propname, buf,len));
 }
 
 /* Convert a device specifier to a fully qualified pathname. */
 ssize_t
 OF_canon(const char *device, char *buf, size_t len)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_CANON(ofw_obj, device, buf, len));
 }
 
 /* Return a package handle for the specified device. */
 phandle_t
 OF_finddevice(const char *device)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_FINDDEVICE(ofw_obj, device));
 }
 
 /* Return the fully qualified pathname corresponding to an instance. */
 ssize_t
 OF_instance_to_path(ihandle_t instance, char *buf, size_t len)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_INSTANCE_TO_PATH(ofw_obj, instance, buf, len));
 }
 
 /* Return the fully qualified pathname corresponding to a package. */
 ssize_t
 OF_package_to_path(phandle_t package, char *buf, size_t len)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_PACKAGE_TO_PATH(ofw_obj, package, buf, len));
 }
 
 /* Look up effective phandle (see FDT/PAPR spec) */
 static phandle_t
 OF_child_xref_phandle(phandle_t parent, phandle_t xref)
 {
 	phandle_t child, rxref;
 
 	/*
 	 * Recursively descend from parent, looking for a node with a property
 	 * named either "phandle", "ibm,phandle", or "linux,phandle" that
 	 * matches the xref we are looking for.
 	 */
 
 	for (child = OF_child(parent); child != 0; child = OF_peer(child)) {
 		rxref = OF_child_xref_phandle(child, xref);
 		if (rxref != -1)
 			return (rxref);
 
 		if (OF_getencprop(child, "phandle", &rxref, sizeof(rxref)) ==
 		    -1 && OF_getencprop(child, "ibm,phandle", &rxref,
 		    sizeof(rxref)) == -1 && OF_getencprop(child,
 		    "linux,phandle", &rxref, sizeof(rxref)) == -1)
 			continue;
 
 		if (rxref == xref)
 			return (child);
 	}
 
 	return (-1);
 }
 
 phandle_t
 OF_node_from_xref(phandle_t xref)
 {
 	struct xrefinfo *xi;
 	phandle_t node;
 
 	if (xref_init_done) {
 		if ((xi = xrefinfo_find(xref, FIND_BY_XREF)) == NULL)
 			return (xref);
 		return (xi->node);
 	}
 
 	if ((node = OF_child_xref_phandle(OF_peer(0), xref)) == -1)
 		return (xref);
 	return (node);
 }
 
 phandle_t
 OF_xref_from_node(phandle_t node)
 {
 	struct xrefinfo *xi;
 	phandle_t xref;
 
 	if (xref_init_done) {
 		if ((xi = xrefinfo_find(node, FIND_BY_NODE)) == NULL)
 			return (node);
 		return (xi->xref);
 	}
 
 	if (OF_getencprop(node, "phandle", &xref, sizeof(xref)) ==
 	    -1 && OF_getencprop(node, "ibm,phandle", &xref,
 	    sizeof(xref)) == -1 && OF_getencprop(node,
 	    "linux,phandle", &xref, sizeof(xref)) == -1)
 		return (node);
 	return (xref);
 }
 
 device_t
 OF_device_from_xref(phandle_t xref)
 {
 	struct xrefinfo *xi;
 
 	if (xref_init_done) {
 		if ((xi = xrefinfo_find(xref, FIND_BY_XREF)) == NULL)
 			return (NULL);
 		return (xi->dev);
 	}
 	panic("Attempt to find device before xreflist_init");
 }
 
 phandle_t
 OF_xref_from_device(device_t dev)
 {
 	struct xrefinfo *xi;
 
 	if (xref_init_done) {
 		if ((xi = xrefinfo_find((uintptr_t)dev, FIND_BY_DEV)) == NULL)
 			return (0);
 		return (xi->xref);
 	}
 	panic("Attempt to find xref before xreflist_init");
 }
 
 int
 OF_device_register_xref(phandle_t xref, device_t dev)
 {
 	struct xrefinfo *xi;
 
 	/*
 	 * If the given xref handle doesn't already exist in the list then we
 	 * add a list entry.  In theory this can only happen on a system where
 	 * nodes don't contain phandle properties and xref and node handles are
 	 * synonymous, so the xref handle is added as the node handle as well.
 	 */
 	if (xref_init_done) {
 		if ((xi = xrefinfo_find(xref, FIND_BY_XREF)) == NULL)
 			xrefinfo_add(xref, xref, dev);
 		else 
 			xi->dev = dev;
 		return (0);
 	}
 	panic("Attempt to register device before xreflist_init");
 }
 
 /*  Call the method in the scope of a given instance. */
 int
 OF_call_method(const char *method, ihandle_t instance, int nargs, int nreturns,
     ...)
 {
 	va_list ap;
 	cell_t args_n_results[12];
 	int n, status;
 
 	if (nargs > 6 || ofw_def_impl == NULL)
 		return (-1);
 	va_start(ap, nreturns);
 	for (n = 0; n < nargs; n++)
 		args_n_results[n] = va_arg(ap, cell_t);
 
 	status = OFW_CALL_METHOD(ofw_obj, instance, method, nargs, nreturns,
 	    args_n_results);
 	if (status != 0)
 		return (status);
 
 	for (; n < nargs + nreturns; n++)
 		*va_arg(ap, cell_t *) = args_n_results[n];
 	va_end(ap);
 	return (0);
 }
 
 /*
  * Device I/O functions
  */
 
 /* Open an instance for a device. */
 ihandle_t
 OF_open(const char *device)
 {
 
 	if (ofw_def_impl == NULL)
 		return (0);
 
 	return (OFW_OPEN(ofw_obj, device));
 }
 
 /* Close an instance. */
 void
 OF_close(ihandle_t instance)
 {
 
 	if (ofw_def_impl == NULL)
 		return;
 
 	OFW_CLOSE(ofw_obj, instance);
 }
 
 /* Read from an instance. */
 ssize_t
 OF_read(ihandle_t instance, void *addr, size_t len)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_READ(ofw_obj, instance, addr, len));
 }
 
 /* Write to an instance. */
 ssize_t
 OF_write(ihandle_t instance, const void *addr, size_t len)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_WRITE(ofw_obj, instance, addr, len));
 }
 
 /* Seek to a position. */
 int
 OF_seek(ihandle_t instance, uint64_t pos)
 {
 
 	if (ofw_def_impl == NULL)
 		return (-1);
 
 	return (OFW_SEEK(ofw_obj, instance, pos));
 }
 
 /*
  * Memory functions
  */
 
 /* Claim an area of memory. */
 void *
 OF_claim(void *virt, size_t size, u_int align)
 {
 
 	if (ofw_def_impl == NULL)
 		return ((void *)-1);
 
 	return (OFW_CLAIM(ofw_obj, virt, size, align));
 }
 
 /* Release an area of memory. */
 void
 OF_release(void *virt, size_t size)
 {
 
 	if (ofw_def_impl == NULL)
 		return;
 
 	OFW_RELEASE(ofw_obj, virt, size);
 }
 
 /*
  * Control transfer functions
  */
 
 /* Suspend and drop back to the Open Firmware interface. */
 void
 OF_enter()
 {
 
 	if (ofw_def_impl == NULL)
 		return;
 
 	OFW_ENTER(ofw_obj);
 }
 
 /* Shut down and drop back to the Open Firmware interface. */
 void
 OF_exit()
 {
 
 	if (ofw_def_impl == NULL)
 		panic("OF_exit: Open Firmware not available");
 
 	/* Should not return */
 	OFW_EXIT(ofw_obj);
 
 	for (;;)			/* just in case */
 		;
 }
Index: projects/clang360-import/sys/libkern/strtoq.c
===================================================================
--- projects/clang360-import/sys/libkern/strtoq.c	(revision 277944)
+++ projects/clang360-import/sys/libkern/strtoq.c	(revision 277945)
@@ -1,130 +1,130 @@
 /*-
  * Copyright (c) 1990, 1993
  *	The Regents of the University of California.  All rights reserved.
  *
  * This code is derived from software contributed to Berkeley by
  * Chris Torek.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  * 4. Neither the name of the University nor the names of its contributors
  *    may be used to endorse or promote products derived from this software
  *    without specific prior written permission.
  *
  * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/ctype.h>
 #include <sys/limits.h>
 
 /*
  * Convert a string to a quad integer.
  *
  * Ignores `locale' stuff.  Assumes that the upper and lower case
  * alphabets and digits are each contiguous.
  */
 quad_t
 strtoq(const char *nptr, char **endptr, int base)
 {
 	const char *s;
 	u_quad_t acc;
 	unsigned char c;
 	u_quad_t qbase, cutoff;
 	int neg, any, cutlim;
 
 	/*
 	 * Skip white space and pick up leading +/- sign if any.
 	 * If base is 0, allow 0x for hex and 0 for octal, else
 	 * assume decimal; if base is already 16, allow 0x.
 	 */
 	s = nptr;
 	do {
 		c = *s++;
 	} while (isspace(c));
 	if (c == '-') {
 		neg = 1;
 		c = *s++;
 	} else {
 		neg = 0;
 		if (c == '+')
 			c = *s++;
 	}
 	if ((base == 0 || base == 16) &&
 	    c == '0' && (*s == 'x' || *s == 'X')) {
 		c = s[1];
 		s += 2;
 		base = 16;
 	}
 	if (base == 0)
 		base = c == '0' ? 8 : 10;
 
 	/*
 	 * Compute the cutoff value between legal numbers and illegal
 	 * numbers.  That is the largest legal value, divided by the
 	 * base.  An input number that is greater than this value, if
 	 * followed by a legal input character, is too big.  One that
 	 * is equal to this value may be valid or not; the limit
 	 * between valid and invalid numbers is then based on the last
 	 * digit.  For instance, if the range for quads is
 	 * [-9223372036854775808..9223372036854775807] and the input base
 	 * is 10, cutoff will be set to 922337203685477580 and cutlim to
 	 * either 7 (neg==0) or 8 (neg==1), meaning that if we have
 	 * accumulated a value > 922337203685477580, or equal but the
 	 * next digit is > 7 (or 8), the number is too big, and we will
 	 * return a range error.
 	 *
 	 * Set any if any `digits' consumed; make it negative to indicate
 	 * overflow.
 	 */
 	qbase = (unsigned)base;
 	cutoff = neg ? (u_quad_t)-(QUAD_MIN + QUAD_MAX) + QUAD_MAX : QUAD_MAX;
 	cutlim = cutoff % qbase;
 	cutoff /= qbase;
 	for (acc = 0, any = 0;; c = *s++) {
 		if (!isascii(c))
 			break;
 		if (isdigit(c))
 			c -= '0';
 		else if (isalpha(c))
 			c -= isupper(c) ? 'A' - 10 : 'a' - 10;
 		else
 			break;
 		if (c >= base)
 			break;
 		if (any < 0 || acc > cutoff || (acc == cutoff && c > cutlim))
 			any = -1;
 		else {
 			any = 1;
 			acc *= qbase;
 			acc += c;
 		}
 	}
 	if (any < 0) {
 		acc = neg ? QUAD_MIN : QUAD_MAX;
 	} else if (neg)
 		acc = -acc;
 	if (endptr != 0)
-		*((const char **)endptr) = any ? s - 1 : nptr;
+		*endptr = __DECONST(char *, any ? s - 1 : nptr);
 	return (acc);
 }
Index: projects/clang360-import/sys/libkern/strtoul.c
===================================================================
--- projects/clang360-import/sys/libkern/strtoul.c	(revision 277944)
+++ projects/clang360-import/sys/libkern/strtoul.c	(revision 277945)
@@ -1,108 +1,108 @@
 /*-
  * Copyright (c) 1990, 1993
  *	The Regents of the University of California.  All rights reserved.
  *
  * This code is derived from software contributed to Berkeley by
  * Chris Torek.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  * 4. Neither the name of the University nor the names of its contributors
  *    may be used to endorse or promote products derived from this software
  *    without specific prior written permission.
  *
  * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  *
  * From: @(#)strtoul.c	8.1 (Berkeley) 6/4/93
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/ctype.h>
 #include <sys/limits.h>
 
 /*
  * Convert a string to an unsigned long integer.
  *
  * Ignores `locale' stuff.  Assumes that the upper and lower case
  * alphabets and digits are each contiguous.
  */
 unsigned long
 strtoul(nptr, endptr, base)
 	const char *nptr;
 	char **endptr;
 	int base;
 {
 	const char *s = nptr;
 	unsigned long acc;
 	unsigned char c;
 	unsigned long cutoff;
 	int neg = 0, any, cutlim;
 
 	/*
 	 * See strtol for comments as to the logic used.
 	 */
 	do {
 		c = *s++;
 	} while (isspace(c));
 	if (c == '-') {
 		neg = 1;
 		c = *s++;
 	} else if (c == '+')
 		c = *s++;
 	if ((base == 0 || base == 16) &&
 	    c == '0' && (*s == 'x' || *s == 'X')) {
 		c = s[1];
 		s += 2;
 		base = 16;
 	}
 	if (base == 0)
 		base = c == '0' ? 8 : 10;
 	cutoff = (unsigned long)ULONG_MAX / (unsigned long)base;
 	cutlim = (unsigned long)ULONG_MAX % (unsigned long)base;
 	for (acc = 0, any = 0;; c = *s++) {
 		if (!isascii(c))
 			break;
 		if (isdigit(c))
 			c -= '0';
 		else if (isalpha(c))
 			c -= isupper(c) ? 'A' - 10 : 'a' - 10;
 		else
 			break;
 		if (c >= base)
 			break;
 		if (any < 0 || acc > cutoff || (acc == cutoff && c > cutlim))
 			any = -1;
 		else {
 			any = 1;
 			acc *= base;
 			acc += c;
 		}
 	}
 	if (any < 0) {
 		acc = ULONG_MAX;
 	} else if (neg)
 		acc = -acc;
 	if (endptr != 0)
-		*((const char **)endptr) = any ? s - 1 : nptr;
+		*endptr = __DECONST(char *, any ? s - 1 : nptr);
 	return (acc);
 }
Index: projects/clang360-import/sys/libkern/strtouq.c
===================================================================
--- projects/clang360-import/sys/libkern/strtouq.c	(revision 277944)
+++ projects/clang360-import/sys/libkern/strtouq.c	(revision 277945)
@@ -1,107 +1,107 @@
 /*-
  * Copyright (c) 1990, 1993
  *	The Regents of the University of California.  All rights reserved.
  *
  * This code is derived from software contributed to Berkeley by
  * Chris Torek.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  * 4. Neither the name of the University nor the names of its contributors
  *    may be used to endorse or promote products derived from this software
  *    without specific prior written permission.
  *
  * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/ctype.h>
 #include <sys/limits.h>
 
 /*
  * Convert a string to an unsigned quad integer.
  *
  * Ignores `locale' stuff.  Assumes that the upper and lower case
  * alphabets and digits are each contiguous.
  */
 u_quad_t
 strtouq(const char *nptr, char **endptr, int base)
 {
 	const char *s = nptr;
 	u_quad_t acc;
 	unsigned char c;
 	u_quad_t qbase, cutoff;
 	int neg, any, cutlim;
 
 	/*
 	 * See strtoq for comments as to the logic used.
 	 */
 	do {
 		c = *s++;
 	} while (isspace(c));
 	if (c == '-') {
 		neg = 1;
 		c = *s++;
 	} else {
 		neg = 0;
 		if (c == '+')
 			c = *s++;
 	}
 	if ((base == 0 || base == 16) &&
 	    c == '0' && (*s == 'x' || *s == 'X')) {
 		c = s[1];
 		s += 2;
 		base = 16;
 	}
 	if (base == 0)
 		base = c == '0' ? 8 : 10;
 	qbase = (unsigned)base;
 	cutoff = (u_quad_t)UQUAD_MAX / qbase;
 	cutlim = (u_quad_t)UQUAD_MAX % qbase;
 	for (acc = 0, any = 0;; c = *s++) {
 		if (!isascii(c))
 			break;
 		if (isdigit(c))
 			c -= '0';
 		else if (isalpha(c))
 			c -= isupper(c) ? 'A' - 10 : 'a' - 10;
 		else
 			break;
 		if (c >= base)
 			break;
 		if (any < 0 || acc > cutoff || (acc == cutoff && c > cutlim))
 			any = -1;
 		else {
 			any = 1;
 			acc *= qbase;
 			acc += c;
 		}
 	}
 	if (any < 0) {
 		acc = UQUAD_MAX;
 	} else if (neg)
 		acc = -acc;
 	if (endptr != 0)
-		*((const char **)endptr) = any ? s - 1 : nptr;
+		*endptr = __DECONST(char *, any ? s - 1 : nptr);
 	return (acc);
 }
Index: projects/clang360-import/sys/modules/aesni/Makefile
===================================================================
--- projects/clang360-import/sys/modules/aesni/Makefile	(revision 277944)
+++ projects/clang360-import/sys/modules/aesni/Makefile	(revision 277945)
@@ -1,24 +1,27 @@
 # $FreeBSD$
 
 .PATH: ${.CURDIR}/../../crypto/aesni
 
 KMOD=	aesni
 SRCS=	aesni.c
 SRCS+=	aeskeys_${MACHINE_CPUARCH}.S
 SRCS+=	device_if.h bus_if.h opt_bus.h cryptodev_if.h
 
 OBJS+=	aesni_ghash.o aesni_wrap.o
 
 # Remove -nostdinc so we can get the intrinsics.
 aesni_ghash.o: aesni_ghash.c
 	# XXX - gcc won't understand -mpclmul
 	${CC} -c ${CFLAGS:C/^-O2$/-O3/:N-nostdinc} ${WERROR} ${PROF} \
 	     -mmmx -msse -msse4 -maes -mpclmul ${.IMPSRC}
 	${CTFCONVERT_CMD}
 
 aesni_wrap.o: aesni_wrap.c
 	${CC} -c ${CFLAGS:C/^-O2$/-O3/:N-nostdinc} ${WERROR} ${PROF} \
 	     -mmmx -msse -msse4 -maes ${.IMPSRC}
 	${CTFCONVERT_CMD}
 
 .include <bsd.kmod.mk>
+
+CWARNFLAGS.aesni_ghash.c=	${NO_WCAST_QUAL}
+CWARNFLAGS.aesni_wrap.c=	${NO_WCAST_QUAL}
Index: projects/clang360-import/sys/netinet/tcp_syncache.c
===================================================================
--- projects/clang360-import/sys/netinet/tcp_syncache.c	(revision 277944)
+++ projects/clang360-import/sys/netinet/tcp_syncache.c	(revision 277945)
@@ -1,2044 +1,2045 @@
 /*-
  * Copyright (c) 2001 McAfee, Inc.
  * Copyright (c) 2006,2013 Andre Oppermann, Internet Business Solutions AG
  * All rights reserved.
  *
  * This software was developed for the FreeBSD Project by Jonathan Lemon
  * and McAfee Research, the Security Research Division of McAfee, Inc. under
  * DARPA/SPAWAR contract N66001-01-C-8035 ("CBOSS"), as part of the
  * DARPA CHATS research program. [2001 McAfee, Inc.]
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHOR AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE AUTHOR OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include "opt_inet.h"
 #include "opt_inet6.h"
 #include "opt_ipsec.h"
 #include "opt_pcbgroup.h"
 
 #include <sys/param.h>
 #include <sys/systm.h>
 #include <sys/kernel.h>
 #include <sys/sysctl.h>
 #include <sys/limits.h>
 #include <sys/lock.h>
 #include <sys/mutex.h>
 #include <sys/malloc.h>
 #include <sys/mbuf.h>
 #include <sys/proc.h>		/* for proc0 declaration */
 #include <sys/random.h>
 #include <sys/socket.h>
 #include <sys/socketvar.h>
 #include <sys/syslog.h>
 #include <sys/ucred.h>
 
 #include <sys/md5.h>
 #include <crypto/siphash/siphash.h>
 
 #include <vm/uma.h>
 
 #include <net/if.h>
 #include <net/if_var.h>
 #include <net/route.h>
 #include <net/vnet.h>
 
 #include <netinet/in.h>
 #include <netinet/in_systm.h>
 #include <netinet/ip.h>
 #include <netinet/in_var.h>
 #include <netinet/in_pcb.h>
 #include <netinet/ip_var.h>
 #include <netinet/ip_options.h>
 #ifdef INET6
 #include <netinet/ip6.h>
 #include <netinet/icmp6.h>
 #include <netinet6/nd6.h>
 #include <netinet6/ip6_var.h>
 #include <netinet6/in6_pcb.h>
 #endif
 #include <netinet/tcp.h>
 #include <netinet/tcp_fsm.h>
 #include <netinet/tcp_seq.h>
 #include <netinet/tcp_timer.h>
 #include <netinet/tcp_var.h>
 #include <netinet/tcp_syncache.h>
 #ifdef INET6
 #include <netinet6/tcp6_var.h>
 #endif
 #ifdef TCP_OFFLOAD
 #include <netinet/toecore.h>
 #endif
 
 #ifdef IPSEC
 #include <netipsec/ipsec.h>
 #ifdef INET6
 #include <netipsec/ipsec6.h>
 #endif
 #include <netipsec/key.h>
 #endif /*IPSEC*/
 
 #include <machine/in_cksum.h>
 
 #include <security/mac/mac_framework.h>
 
 static VNET_DEFINE(int, tcp_syncookies) = 1;
 #define	V_tcp_syncookies		VNET(tcp_syncookies)
 SYSCTL_INT(_net_inet_tcp, OID_AUTO, syncookies, CTLFLAG_VNET | CTLFLAG_RW,
     &VNET_NAME(tcp_syncookies), 0,
     "Use TCP SYN cookies if the syncache overflows");
 
 static VNET_DEFINE(int, tcp_syncookiesonly) = 0;
 #define	V_tcp_syncookiesonly		VNET(tcp_syncookiesonly)
 SYSCTL_INT(_net_inet_tcp, OID_AUTO, syncookies_only, CTLFLAG_VNET | CTLFLAG_RW,
     &VNET_NAME(tcp_syncookiesonly), 0,
     "Use only TCP SYN cookies");
 
 #ifdef TCP_OFFLOAD
 #define ADDED_BY_TOE(sc) ((sc)->sc_tod != NULL)
 #endif
 
 static void	 syncache_drop(struct syncache *, struct syncache_head *);
 static void	 syncache_free(struct syncache *);
 static void	 syncache_insert(struct syncache *, struct syncache_head *);
 static int	 syncache_respond(struct syncache *, struct syncache_head *, int);
 static struct	 socket *syncache_socket(struct syncache *, struct socket *,
 		    struct mbuf *m);
 static void	 syncache_timeout(struct syncache *sc, struct syncache_head *sch,
 		    int docallout);
 static void	 syncache_timer(void *);
 
 static uint32_t	 syncookie_mac(struct in_conninfo *, tcp_seq, uint8_t,
 		    uint8_t *, uintptr_t);
 static tcp_seq	 syncookie_generate(struct syncache_head *, struct syncache *);
 static struct syncache
 		*syncookie_lookup(struct in_conninfo *, struct syncache_head *,
 		    struct syncache *, struct tcphdr *, struct tcpopt *,
 		    struct socket *);
 static void	 syncookie_reseed(void *);
 #ifdef INVARIANTS
 static int	 syncookie_cmp(struct in_conninfo *inc, struct syncache_head *sch,
 		    struct syncache *sc, struct tcphdr *th, struct tcpopt *to,
 		    struct socket *lso);
 #endif
 
 /*
  * Transmit the SYN,ACK fewer times than TCP_MAXRXTSHIFT specifies.
  * 3 retransmits corresponds to a timeout of 3 * (1 + 2 + 4 + 8) == 45 seconds,
  * the odds are that the user has given up attempting to connect by then.
  */
 #define SYNCACHE_MAXREXMTS		3
 
 /* Arbitrary values */
 #define TCP_SYNCACHE_HASHSIZE		512
 #define TCP_SYNCACHE_BUCKETLIMIT	30
 
 static VNET_DEFINE(struct tcp_syncache, tcp_syncache);
 #define	V_tcp_syncache			VNET(tcp_syncache)
 
 static SYSCTL_NODE(_net_inet_tcp, OID_AUTO, syncache, CTLFLAG_RW, 0,
     "TCP SYN cache");
 
 SYSCTL_UINT(_net_inet_tcp_syncache, OID_AUTO, bucketlimit, CTLFLAG_VNET | CTLFLAG_RDTUN,
     &VNET_NAME(tcp_syncache.bucket_limit), 0,
     "Per-bucket hash limit for syncache");
 
 SYSCTL_UINT(_net_inet_tcp_syncache, OID_AUTO, cachelimit, CTLFLAG_VNET | CTLFLAG_RDTUN,
     &VNET_NAME(tcp_syncache.cache_limit), 0,
     "Overall entry limit for syncache");
 
 SYSCTL_UMA_CUR(_net_inet_tcp_syncache, OID_AUTO, count, CTLFLAG_VNET,
     &VNET_NAME(tcp_syncache.zone), "Current number of entries in syncache");
 
 SYSCTL_UINT(_net_inet_tcp_syncache, OID_AUTO, hashsize, CTLFLAG_VNET | CTLFLAG_RDTUN,
     &VNET_NAME(tcp_syncache.hashsize), 0,
     "Size of TCP syncache hashtable");
 
 SYSCTL_UINT(_net_inet_tcp_syncache, OID_AUTO, rexmtlimit, CTLFLAG_VNET | CTLFLAG_RW,
     &VNET_NAME(tcp_syncache.rexmt_limit), 0,
     "Limit on SYN/ACK retransmissions");
 
 VNET_DEFINE(int, tcp_sc_rst_sock_fail) = 1;
 SYSCTL_INT(_net_inet_tcp_syncache, OID_AUTO, rst_on_sock_fail,
     CTLFLAG_VNET | CTLFLAG_RW, &VNET_NAME(tcp_sc_rst_sock_fail), 0,
     "Send reset on socket allocation failure");
 
 static MALLOC_DEFINE(M_SYNCACHE, "syncache", "TCP syncache");
 
 #define SYNCACHE_HASH(inc, mask)					\
 	((V_tcp_syncache.hash_secret ^					\
 	  (inc)->inc_faddr.s_addr ^					\
 	  ((inc)->inc_faddr.s_addr >> 16) ^				\
 	  (inc)->inc_fport ^ (inc)->inc_lport) & mask)
 
 #define SYNCACHE_HASH6(inc, mask)					\
 	((V_tcp_syncache.hash_secret ^					\
 	  (inc)->inc6_faddr.s6_addr32[0] ^				\
 	  (inc)->inc6_faddr.s6_addr32[3] ^				\
 	  (inc)->inc_fport ^ (inc)->inc_lport) & mask)
 
 #define ENDPTS_EQ(a, b) (						\
 	(a)->ie_fport == (b)->ie_fport &&				\
 	(a)->ie_lport == (b)->ie_lport &&				\
 	(a)->ie_faddr.s_addr == (b)->ie_faddr.s_addr &&			\
 	(a)->ie_laddr.s_addr == (b)->ie_laddr.s_addr			\
 )
 
 #define ENDPTS6_EQ(a, b) (memcmp(a, b, sizeof(*a)) == 0)
 
 #define	SCH_LOCK(sch)		mtx_lock(&(sch)->sch_mtx)
 #define	SCH_UNLOCK(sch)		mtx_unlock(&(sch)->sch_mtx)
 #define	SCH_LOCK_ASSERT(sch)	mtx_assert(&(sch)->sch_mtx, MA_OWNED)
 
 /*
  * Requires the syncache entry to be already removed from the bucket list.
  */
 static void
 syncache_free(struct syncache *sc)
 {
 
 	if (sc->sc_ipopts)
 		(void) m_free(sc->sc_ipopts);
 	if (sc->sc_cred)
 		crfree(sc->sc_cred);
 #ifdef MAC
 	mac_syncache_destroy(&sc->sc_label);
 #endif
 
 	uma_zfree(V_tcp_syncache.zone, sc);
 }
 
 void
 syncache_init(void)
 {
 	int i;
 
 	V_tcp_syncache.hashsize = TCP_SYNCACHE_HASHSIZE;
 	V_tcp_syncache.bucket_limit = TCP_SYNCACHE_BUCKETLIMIT;
 	V_tcp_syncache.rexmt_limit = SYNCACHE_MAXREXMTS;
 	V_tcp_syncache.hash_secret = arc4random();
 
 	TUNABLE_INT_FETCH("net.inet.tcp.syncache.hashsize",
 	    &V_tcp_syncache.hashsize);
 	TUNABLE_INT_FETCH("net.inet.tcp.syncache.bucketlimit",
 	    &V_tcp_syncache.bucket_limit);
 	if (!powerof2(V_tcp_syncache.hashsize) ||
 	    V_tcp_syncache.hashsize == 0) {
 		printf("WARNING: syncache hash size is not a power of 2.\n");
 		V_tcp_syncache.hashsize = TCP_SYNCACHE_HASHSIZE;
 	}
 	V_tcp_syncache.hashmask = V_tcp_syncache.hashsize - 1;
 
 	/* Set limits. */
 	V_tcp_syncache.cache_limit =
 	    V_tcp_syncache.hashsize * V_tcp_syncache.bucket_limit;
 	TUNABLE_INT_FETCH("net.inet.tcp.syncache.cachelimit",
 	    &V_tcp_syncache.cache_limit);
 
 	/* Allocate the hash table. */
 	V_tcp_syncache.hashbase = malloc(V_tcp_syncache.hashsize *
 	    sizeof(struct syncache_head), M_SYNCACHE, M_WAITOK | M_ZERO);
 
 #ifdef VIMAGE
 	V_tcp_syncache.vnet = curvnet;
 #endif
 
 	/* Initialize the hash buckets. */
 	for (i = 0; i < V_tcp_syncache.hashsize; i++) {
 		TAILQ_INIT(&V_tcp_syncache.hashbase[i].sch_bucket);
 		mtx_init(&V_tcp_syncache.hashbase[i].sch_mtx, "tcp_sc_head",
 			 NULL, MTX_DEF);
 		callout_init_mtx(&V_tcp_syncache.hashbase[i].sch_timer,
 			 &V_tcp_syncache.hashbase[i].sch_mtx, 0);
 		V_tcp_syncache.hashbase[i].sch_length = 0;
 		V_tcp_syncache.hashbase[i].sch_sc = &V_tcp_syncache;
 	}
 
 	/* Create the syncache entry zone. */
 	V_tcp_syncache.zone = uma_zcreate("syncache", sizeof(struct syncache),
 	    NULL, NULL, NULL, NULL, UMA_ALIGN_PTR, 0);
 	V_tcp_syncache.cache_limit = uma_zone_set_max(V_tcp_syncache.zone,
 	    V_tcp_syncache.cache_limit);
 
 	/* Start the SYN cookie reseeder callout. */
 	callout_init(&V_tcp_syncache.secret.reseed, 1);
 	arc4rand(V_tcp_syncache.secret.key[0], SYNCOOKIE_SECRET_SIZE, 0);
 	arc4rand(V_tcp_syncache.secret.key[1], SYNCOOKIE_SECRET_SIZE, 0);
 	callout_reset(&V_tcp_syncache.secret.reseed, SYNCOOKIE_LIFETIME * hz,
 	    syncookie_reseed, &V_tcp_syncache);
 }
 
 #ifdef VIMAGE
 void
 syncache_destroy(void)
 {
 	struct syncache_head *sch;
 	struct syncache *sc, *nsc;
 	int i;
 
 	/* Cleanup hash buckets: stop timers, free entries, destroy locks. */
 	for (i = 0; i < V_tcp_syncache.hashsize; i++) {
 
 		sch = &V_tcp_syncache.hashbase[i];
 		callout_drain(&sch->sch_timer);
 
 		SCH_LOCK(sch);
 		TAILQ_FOREACH_SAFE(sc, &sch->sch_bucket, sc_hash, nsc)
 			syncache_drop(sc, sch);
 		SCH_UNLOCK(sch);
 		KASSERT(TAILQ_EMPTY(&sch->sch_bucket),
 		    ("%s: sch->sch_bucket not empty", __func__));
 		KASSERT(sch->sch_length == 0, ("%s: sch->sch_length %d not 0",
 		    __func__, sch->sch_length));
 		mtx_destroy(&sch->sch_mtx);
 	}
 
 	KASSERT(uma_zone_get_cur(V_tcp_syncache.zone) == 0,
 	    ("%s: cache_count not 0", __func__));
 
 	/* Free the allocated global resources. */
 	uma_zdestroy(V_tcp_syncache.zone);
 	free(V_tcp_syncache.hashbase, M_SYNCACHE);
 
 	callout_drain(&V_tcp_syncache.secret.reseed);
 }
 #endif
 
 /*
  * Inserts a syncache entry into the specified bucket row.
  * Locks and unlocks the syncache_head autonomously.
  */
 static void
 syncache_insert(struct syncache *sc, struct syncache_head *sch)
 {
 	struct syncache *sc2;
 
 	SCH_LOCK(sch);
 
 	/*
 	 * Make sure that we don't overflow the per-bucket limit.
 	 * If the bucket is full, toss the oldest element.
 	 */
 	if (sch->sch_length >= V_tcp_syncache.bucket_limit) {
 		KASSERT(!TAILQ_EMPTY(&sch->sch_bucket),
 			("sch->sch_length incorrect"));
 		sc2 = TAILQ_LAST(&sch->sch_bucket, sch_head);
 		syncache_drop(sc2, sch);
 		TCPSTAT_INC(tcps_sc_bucketoverflow);
 	}
 
 	/* Put it into the bucket. */
 	TAILQ_INSERT_HEAD(&sch->sch_bucket, sc, sc_hash);
 	sch->sch_length++;
 
 #ifdef TCP_OFFLOAD
 	if (ADDED_BY_TOE(sc)) {
 		struct toedev *tod = sc->sc_tod;
 
 		tod->tod_syncache_added(tod, sc->sc_todctx);
 	}
 #endif
 
 	/* Reinitialize the bucket row's timer. */
 	if (sch->sch_length == 1)
 		sch->sch_nextc = ticks + INT_MAX;
 	syncache_timeout(sc, sch, 1);
 
 	SCH_UNLOCK(sch);
 
 	TCPSTAT_INC(tcps_sc_added);
 }
 
 /*
  * Remove and free entry from syncache bucket row.
  * Expects locked syncache head.
  */
 static void
 syncache_drop(struct syncache *sc, struct syncache_head *sch)
 {
 
 	SCH_LOCK_ASSERT(sch);
 
 	TAILQ_REMOVE(&sch->sch_bucket, sc, sc_hash);
 	sch->sch_length--;
 
 #ifdef TCP_OFFLOAD
 	if (ADDED_BY_TOE(sc)) {
 		struct toedev *tod = sc->sc_tod;
 
 		tod->tod_syncache_removed(tod, sc->sc_todctx);
 	}
 #endif
 
 	syncache_free(sc);
 }
 
 /*
  * Engage/reengage time on bucket row.
  */
 static void
 syncache_timeout(struct syncache *sc, struct syncache_head *sch, int docallout)
 {
 	sc->sc_rxttime = ticks +
 		TCPTV_RTOBASE * (tcp_syn_backoff[sc->sc_rxmits]);
 	sc->sc_rxmits++;
 	if (TSTMP_LT(sc->sc_rxttime, sch->sch_nextc)) {
 		sch->sch_nextc = sc->sc_rxttime;
 		if (docallout)
 			callout_reset(&sch->sch_timer, sch->sch_nextc - ticks,
 			    syncache_timer, (void *)sch);
 	}
 }
 
 /*
  * Walk the timer queues, looking for SYN,ACKs that need to be retransmitted.
  * If we have retransmitted an entry the maximum number of times, expire it.
  * One separate timer for each bucket row.
  */
 static void
 syncache_timer(void *xsch)
 {
 	struct syncache_head *sch = (struct syncache_head *)xsch;
 	struct syncache *sc, *nsc;
 	int tick = ticks;
 	char *s;
 
 	CURVNET_SET(sch->sch_sc->vnet);
 
 	/* NB: syncache_head has already been locked by the callout. */
 	SCH_LOCK_ASSERT(sch);
 
 	/*
 	 * In the following cycle we may remove some entries and/or
 	 * advance some timeouts, so re-initialize the bucket timer.
 	 */
 	sch->sch_nextc = tick + INT_MAX;
 
 	TAILQ_FOREACH_SAFE(sc, &sch->sch_bucket, sc_hash, nsc) {
 		/*
 		 * We do not check if the listen socket still exists
 		 * and accept the case where the listen socket may be
 		 * gone by the time we resend the SYN/ACK.  We do
 		 * not expect this to happens often. If it does,
 		 * then the RST will be sent by the time the remote
 		 * host does the SYN/ACK->ACK.
 		 */
 		if (TSTMP_GT(sc->sc_rxttime, tick)) {
 			if (TSTMP_LT(sc->sc_rxttime, sch->sch_nextc))
 				sch->sch_nextc = sc->sc_rxttime;
 			continue;
 		}
 		if (sc->sc_rxmits > V_tcp_syncache.rexmt_limit) {
 			if ((s = tcp_log_addrs(&sc->sc_inc, NULL, NULL, NULL))) {
 				log(LOG_DEBUG, "%s; %s: Retransmits exhausted, "
 				    "giving up and removing syncache entry\n",
 				    s, __func__);
 				free(s, M_TCPLOG);
 			}
 			syncache_drop(sc, sch);
 			TCPSTAT_INC(tcps_sc_stale);
 			continue;
 		}
 		if ((s = tcp_log_addrs(&sc->sc_inc, NULL, NULL, NULL))) {
 			log(LOG_DEBUG, "%s; %s: Response timeout, "
 			    "retransmitting (%u) SYN|ACK\n",
 			    s, __func__, sc->sc_rxmits);
 			free(s, M_TCPLOG);
 		}
 
 		syncache_respond(sc, sch, 1);
 		TCPSTAT_INC(tcps_sc_retransmitted);
 		syncache_timeout(sc, sch, 0);
 	}
 	if (!TAILQ_EMPTY(&(sch)->sch_bucket))
 		callout_reset(&(sch)->sch_timer, (sch)->sch_nextc - tick,
 			syncache_timer, (void *)(sch));
 	CURVNET_RESTORE();
 }
 
 /*
  * Find an entry in the syncache.
  * Returns always with locked syncache_head plus a matching entry or NULL.
  */
 static struct syncache *
 syncache_lookup(struct in_conninfo *inc, struct syncache_head **schp)
 {
 	struct syncache *sc;
 	struct syncache_head *sch;
 
 #ifdef INET6
 	if (inc->inc_flags & INC_ISIPV6) {
 		sch = &V_tcp_syncache.hashbase[
 		    SYNCACHE_HASH6(inc, V_tcp_syncache.hashmask)];
 		*schp = sch;
 
 		SCH_LOCK(sch);
 
 		/* Circle through bucket row to find matching entry. */
 		TAILQ_FOREACH(sc, &sch->sch_bucket, sc_hash) {
 			if (ENDPTS6_EQ(&inc->inc_ie, &sc->sc_inc.inc_ie))
 				return (sc);
 		}
 	} else
 #endif
 	{
 		sch = &V_tcp_syncache.hashbase[
 		    SYNCACHE_HASH(inc, V_tcp_syncache.hashmask)];
 		*schp = sch;
 
 		SCH_LOCK(sch);
 
 		/* Circle through bucket row to find matching entry. */
 		TAILQ_FOREACH(sc, &sch->sch_bucket, sc_hash) {
 #ifdef INET6
 			if (sc->sc_inc.inc_flags & INC_ISIPV6)
 				continue;
 #endif
 			if (ENDPTS_EQ(&inc->inc_ie, &sc->sc_inc.inc_ie))
 				return (sc);
 		}
 	}
 	SCH_LOCK_ASSERT(*schp);
 	return (NULL);			/* always returns with locked sch */
 }
 
 /*
  * This function is called when we get a RST for a
  * non-existent connection, so that we can see if the
  * connection is in the syn cache.  If it is, zap it.
  */
 void
 syncache_chkrst(struct in_conninfo *inc, struct tcphdr *th)
 {
 	struct syncache *sc;
 	struct syncache_head *sch;
 	char *s = NULL;
 
 	sc = syncache_lookup(inc, &sch);	/* returns locked sch */
 	SCH_LOCK_ASSERT(sch);
 
 	/*
 	 * Any RST to our SYN|ACK must not carry ACK, SYN or FIN flags.
 	 * See RFC 793 page 65, section SEGMENT ARRIVES.
 	 */
 	if (th->th_flags & (TH_ACK|TH_SYN|TH_FIN)) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: Spurious RST with ACK, SYN or "
 			    "FIN flag set, segment ignored\n", s, __func__);
 		TCPSTAT_INC(tcps_badrst);
 		goto done;
 	}
 
 	/*
 	 * No corresponding connection was found in syncache.
 	 * If syncookies are enabled and possibly exclusively
 	 * used, or we are under memory pressure, a valid RST
 	 * may not find a syncache entry.  In that case we're
 	 * done and no SYN|ACK retransmissions will happen.
 	 * Otherwise the RST was misdirected or spoofed.
 	 */
 	if (sc == NULL) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: Spurious RST without matching "
 			    "syncache entry (possibly syncookie only), "
 			    "segment ignored\n", s, __func__);
 		TCPSTAT_INC(tcps_badrst);
 		goto done;
 	}
 
 	/*
 	 * If the RST bit is set, check the sequence number to see
 	 * if this is a valid reset segment.
 	 * RFC 793 page 37:
 	 *   In all states except SYN-SENT, all reset (RST) segments
 	 *   are validated by checking their SEQ-fields.  A reset is
 	 *   valid if its sequence number is in the window.
 	 *
 	 *   The sequence number in the reset segment is normally an
 	 *   echo of our outgoing acknowlegement numbers, but some hosts
 	 *   send a reset with the sequence number at the rightmost edge
 	 *   of our receive window, and we have to handle this case.
 	 */
 	if (SEQ_GEQ(th->th_seq, sc->sc_irs) &&
 	    SEQ_LEQ(th->th_seq, sc->sc_irs + sc->sc_wnd)) {
 		syncache_drop(sc, sch);
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: Our SYN|ACK was rejected, "
 			    "connection attempt aborted by remote endpoint\n",
 			    s, __func__);
 		TCPSTAT_INC(tcps_sc_reset);
 	} else {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: RST with invalid SEQ %u != "
 			    "IRS %u (+WND %u), segment ignored\n",
 			    s, __func__, th->th_seq, sc->sc_irs, sc->sc_wnd);
 		TCPSTAT_INC(tcps_badrst);
 	}
 
 done:
 	if (s != NULL)
 		free(s, M_TCPLOG);
 	SCH_UNLOCK(sch);
 }
 
 void
 syncache_badack(struct in_conninfo *inc)
 {
 	struct syncache *sc;
 	struct syncache_head *sch;
 
 	sc = syncache_lookup(inc, &sch);	/* returns locked sch */
 	SCH_LOCK_ASSERT(sch);
 	if (sc != NULL) {
 		syncache_drop(sc, sch);
 		TCPSTAT_INC(tcps_sc_badack);
 	}
 	SCH_UNLOCK(sch);
 }
 
 void
 syncache_unreach(struct in_conninfo *inc, struct tcphdr *th)
 {
 	struct syncache *sc;
 	struct syncache_head *sch;
 
 	sc = syncache_lookup(inc, &sch);	/* returns locked sch */
 	SCH_LOCK_ASSERT(sch);
 	if (sc == NULL)
 		goto done;
 
 	/* If the sequence number != sc_iss, then it's a bogus ICMP msg */
 	if (ntohl(th->th_seq) != sc->sc_iss)
 		goto done;
 
 	/*
 	 * If we've rertransmitted 3 times and this is our second error,
 	 * we remove the entry.  Otherwise, we allow it to continue on.
 	 * This prevents us from incorrectly nuking an entry during a
 	 * spurious network outage.
 	 *
 	 * See tcp_notify().
 	 */
 	if ((sc->sc_flags & SCF_UNREACH) == 0 || sc->sc_rxmits < 3 + 1) {
 		sc->sc_flags |= SCF_UNREACH;
 		goto done;
 	}
 	syncache_drop(sc, sch);
 	TCPSTAT_INC(tcps_sc_unreach);
 done:
 	SCH_UNLOCK(sch);
 }
 
 /*
  * Build a new TCP socket structure from a syncache entry.
  */
 static struct socket *
 syncache_socket(struct syncache *sc, struct socket *lso, struct mbuf *m)
 {
 	struct inpcb *inp = NULL;
 	struct socket *so;
 	struct tcpcb *tp;
 	int error;
 	char *s;
 
 	INP_INFO_WLOCK_ASSERT(&V_tcbinfo);
 
 	/*
 	 * Ok, create the full blown connection, and set things up
 	 * as they would have been set up if we had created the
 	 * connection when the SYN arrived.  If we can't create
 	 * the connection, abort it.
 	 */
 	so = sonewconn(lso, 0);
 	if (so == NULL) {
 		/*
 		 * Drop the connection; we will either send a RST or
 		 * have the peer retransmit its SYN again after its
 		 * RTO and try again.
 		 */
 		TCPSTAT_INC(tcps_listendrop);
 		if ((s = tcp_log_addrs(&sc->sc_inc, NULL, NULL, NULL))) {
 			log(LOG_DEBUG, "%s; %s: Socket create failed "
 			    "due to limits or memory shortage\n",
 			    s, __func__);
 			free(s, M_TCPLOG);
 		}
 		goto abort2;
 	}
 #ifdef MAC
 	mac_socketpeer_set_from_mbuf(m, so);
 #endif
 
 	inp = sotoinpcb(so);
 	inp->inp_inc.inc_fibnum = so->so_fibnum;
 	INP_WLOCK(inp);
 	INP_HASH_WLOCK(&V_tcbinfo);
 
 	/* Insert new socket into PCB hash list. */
 	inp->inp_inc.inc_flags = sc->sc_inc.inc_flags;
 #ifdef INET6
 	if (sc->sc_inc.inc_flags & INC_ISIPV6) {
 		inp->in6p_laddr = sc->sc_inc.inc6_laddr;
 	} else {
 		inp->inp_vflag &= ~INP_IPV6;
 		inp->inp_vflag |= INP_IPV4;
 #endif
 		inp->inp_laddr = sc->sc_inc.inc_laddr;
 #ifdef INET6
 	}
 #endif
 
 	/*
 	 * If there's an mbuf and it has a flowid, then let's initialise the
 	 * inp with that particular flowid.
 	 */
 	if (m != NULL && M_HASHTYPE_GET(m) != M_HASHTYPE_NONE) {
 		inp->inp_flowid = m->m_pkthdr.flowid;
 		inp->inp_flowtype = M_HASHTYPE_GET(m);
 	}
 
 	/*
 	 * Install in the reservation hash table for now, but don't yet
 	 * install a connection group since the full 4-tuple isn't yet
 	 * configured.
 	 */
 	inp->inp_lport = sc->sc_inc.inc_lport;
 	if ((error = in_pcbinshash_nopcbgroup(inp)) != 0) {
 		/*
 		 * Undo the assignments above if we failed to
 		 * put the PCB on the hash lists.
 		 */
 #ifdef INET6
 		if (sc->sc_inc.inc_flags & INC_ISIPV6)
 			inp->in6p_laddr = in6addr_any;
 		else
 #endif
 			inp->inp_laddr.s_addr = INADDR_ANY;
 		inp->inp_lport = 0;
 		if ((s = tcp_log_addrs(&sc->sc_inc, NULL, NULL, NULL))) {
 			log(LOG_DEBUG, "%s; %s: in_pcbinshash failed "
 			    "with error %i\n",
 			    s, __func__, error);
 			free(s, M_TCPLOG);
 		}
 		INP_HASH_WUNLOCK(&V_tcbinfo);
 		goto abort;
 	}
 #ifdef IPSEC
 	/* Copy old policy into new socket's. */
 	if (ipsec_copy_policy(sotoinpcb(lso)->inp_sp, inp->inp_sp))
 		printf("syncache_socket: could not copy policy\n");
 #endif
 #ifdef INET6
 	if (sc->sc_inc.inc_flags & INC_ISIPV6) {
 		struct inpcb *oinp = sotoinpcb(lso);
 		struct in6_addr laddr6;
 		struct sockaddr_in6 sin6;
 		/*
 		 * Inherit socket options from the listening socket.
 		 * Note that in6p_inputopts are not (and should not be)
 		 * copied, since it stores previously received options and is
 		 * used to detect if each new option is different than the
 		 * previous one and hence should be passed to a user.
 		 * If we copied in6p_inputopts, a user would not be able to
 		 * receive options just after calling the accept system call.
 		 */
 		inp->inp_flags |= oinp->inp_flags & INP_CONTROLOPTS;
 		if (oinp->in6p_outputopts)
 			inp->in6p_outputopts =
 			    ip6_copypktopts(oinp->in6p_outputopts, M_NOWAIT);
 
 		sin6.sin6_family = AF_INET6;
 		sin6.sin6_len = sizeof(sin6);
 		sin6.sin6_addr = sc->sc_inc.inc6_faddr;
 		sin6.sin6_port = sc->sc_inc.inc_fport;
 		sin6.sin6_flowinfo = sin6.sin6_scope_id = 0;
 		laddr6 = inp->in6p_laddr;
 		if (IN6_IS_ADDR_UNSPECIFIED(&inp->in6p_laddr))
 			inp->in6p_laddr = sc->sc_inc.inc6_laddr;
 		if ((error = in6_pcbconnect_mbuf(inp, (struct sockaddr *)&sin6,
 		    thread0.td_ucred, m)) != 0) {
 			inp->in6p_laddr = laddr6;
 			if ((s = tcp_log_addrs(&sc->sc_inc, NULL, NULL, NULL))) {
 				log(LOG_DEBUG, "%s; %s: in6_pcbconnect failed "
 				    "with error %i\n",
 				    s, __func__, error);
 				free(s, M_TCPLOG);
 			}
 			INP_HASH_WUNLOCK(&V_tcbinfo);
 			goto abort;
 		}
 		/* Override flowlabel from in6_pcbconnect. */
 		inp->inp_flow &= ~IPV6_FLOWLABEL_MASK;
 		inp->inp_flow |= sc->sc_flowlabel;
 	}
 #endif /* INET6 */
 #if defined(INET) && defined(INET6)
 	else
 #endif
 #ifdef INET
 	{
 		struct in_addr laddr;
 		struct sockaddr_in sin;
 
 		inp->inp_options = (m) ? ip_srcroute(m) : NULL;
 		
 		if (inp->inp_options == NULL) {
 			inp->inp_options = sc->sc_ipopts;
 			sc->sc_ipopts = NULL;
 		}
 
 		sin.sin_family = AF_INET;
 		sin.sin_len = sizeof(sin);
 		sin.sin_addr = sc->sc_inc.inc_faddr;
 		sin.sin_port = sc->sc_inc.inc_fport;
 		bzero((caddr_t)sin.sin_zero, sizeof(sin.sin_zero));
 		laddr = inp->inp_laddr;
 		if (inp->inp_laddr.s_addr == INADDR_ANY)
 			inp->inp_laddr = sc->sc_inc.inc_laddr;
 		if ((error = in_pcbconnect_mbuf(inp, (struct sockaddr *)&sin,
 		    thread0.td_ucred, m)) != 0) {
 			inp->inp_laddr = laddr;
 			if ((s = tcp_log_addrs(&sc->sc_inc, NULL, NULL, NULL))) {
 				log(LOG_DEBUG, "%s; %s: in_pcbconnect failed "
 				    "with error %i\n",
 				    s, __func__, error);
 				free(s, M_TCPLOG);
 			}
 			INP_HASH_WUNLOCK(&V_tcbinfo);
 			goto abort;
 		}
 	}
 #endif /* INET */
 	INP_HASH_WUNLOCK(&V_tcbinfo);
 	tp = intotcpcb(inp);
 	tcp_state_change(tp, TCPS_SYN_RECEIVED);
 	tp->iss = sc->sc_iss;
 	tp->irs = sc->sc_irs;
 	tcp_rcvseqinit(tp);
 	tcp_sendseqinit(tp);
 	tp->snd_wl1 = sc->sc_irs;
 	tp->snd_max = tp->iss + 1;
 	tp->snd_nxt = tp->iss + 1;
 	tp->rcv_up = sc->sc_irs + 1;
 	tp->rcv_wnd = sc->sc_wnd;
 	tp->rcv_adv += tp->rcv_wnd;
 	tp->last_ack_sent = tp->rcv_nxt;
 
 	tp->t_flags = sototcpcb(lso)->t_flags & (TF_NOPUSH|TF_NODELAY);
 	if (sc->sc_flags & SCF_NOOPT)
 		tp->t_flags |= TF_NOOPT;
 	else {
 		if (sc->sc_flags & SCF_WINSCALE) {
 			tp->t_flags |= TF_REQ_SCALE|TF_RCVD_SCALE;
 			tp->snd_scale = sc->sc_requested_s_scale;
 			tp->request_r_scale = sc->sc_requested_r_scale;
 		}
 		if (sc->sc_flags & SCF_TIMESTAMP) {
 			tp->t_flags |= TF_REQ_TSTMP|TF_RCVD_TSTMP;
 			tp->ts_recent = sc->sc_tsreflect;
 			tp->ts_recent_age = tcp_ts_getticks();
 			tp->ts_offset = sc->sc_tsoff;
 		}
 #ifdef TCP_SIGNATURE
 		if (sc->sc_flags & SCF_SIGNATURE)
 			tp->t_flags |= TF_SIGNATURE;
 #endif
 		if (sc->sc_flags & SCF_SACK)
 			tp->t_flags |= TF_SACK_PERMIT;
 	}
 
 	if (sc->sc_flags & SCF_ECN)
 		tp->t_flags |= TF_ECN_PERMIT;
 
 	/*
 	 * Set up MSS and get cached values from tcp_hostcache.
 	 * This might overwrite some of the defaults we just set.
 	 */
 	tcp_mss(tp, sc->sc_peer_mss);
 
 	/*
 	 * If the SYN,ACK was retransmitted, indicate that CWND to be
 	 * limited to one segment in cc_conn_init().
 	 * NB: sc_rxmits counts all SYN,ACK transmits, not just retransmits.
 	 */
 	if (sc->sc_rxmits > 1)
 		tp->snd_cwnd = 1;
 
 #ifdef TCP_OFFLOAD
 	/*
 	 * Allow a TOE driver to install its hooks.  Note that we hold the
 	 * pcbinfo lock too and that prevents tcp_usr_accept from accepting a
 	 * new connection before the TOE driver has done its thing.
 	 */
 	if (ADDED_BY_TOE(sc)) {
 		struct toedev *tod = sc->sc_tod;
 
 		tod->tod_offload_socket(tod, sc->sc_todctx, so);
 	}
 #endif
 	/*
 	 * Copy and activate timers.
 	 */
 	tp->t_keepinit = sototcpcb(lso)->t_keepinit;
 	tp->t_keepidle = sototcpcb(lso)->t_keepidle;
 	tp->t_keepintvl = sototcpcb(lso)->t_keepintvl;
 	tp->t_keepcnt = sototcpcb(lso)->t_keepcnt;
 	tcp_timer_activate(tp, TT_KEEP, TP_KEEPINIT(tp));
 
 	INP_WUNLOCK(inp);
 
 	soisconnected(so);
 
 	TCPSTAT_INC(tcps_accepts);
 	return (so);
 
 abort:
 	INP_WUNLOCK(inp);
 abort2:
 	if (so != NULL)
 		soabort(so);
 	return (NULL);
 }
 
 /*
  * This function gets called when we receive an ACK for a
  * socket in the LISTEN state.  We look up the connection
  * in the syncache, and if its there, we pull it out of
  * the cache and turn it into a full-blown connection in
  * the SYN-RECEIVED state.
  */
 int
 syncache_expand(struct in_conninfo *inc, struct tcpopt *to, struct tcphdr *th,
     struct socket **lsop, struct mbuf *m)
 {
 	struct syncache *sc;
 	struct syncache_head *sch;
 	struct syncache scs;
 	char *s;
 
 	/*
 	 * Global TCP locks are held because we manipulate the PCB lists
 	 * and create a new socket.
 	 */
 	INP_INFO_WLOCK_ASSERT(&V_tcbinfo);
 	KASSERT((th->th_flags & (TH_RST|TH_ACK|TH_SYN)) == TH_ACK,
 	    ("%s: can handle only ACK", __func__));
 
 	sc = syncache_lookup(inc, &sch);	/* returns locked sch */
 	SCH_LOCK_ASSERT(sch);
 
 #ifdef INVARIANTS
 	/*
 	 * Test code for syncookies comparing the syncache stored
 	 * values with the reconstructed values from the cookie.
 	 */
 	if (sc != NULL)
 		syncookie_cmp(inc, sch, sc, th, to, *lsop);
 #endif
 
 	if (sc == NULL) {
 		/*
 		 * There is no syncache entry, so see if this ACK is
 		 * a returning syncookie.  To do this, first:
 		 *  A. See if this socket has had a syncache entry dropped in
 		 *     the past.  We don't want to accept a bogus syncookie
 		 *     if we've never received a SYN.
 		 *  B. check that the syncookie is valid.  If it is, then
 		 *     cobble up a fake syncache entry, and return.
 		 */
 		if (!V_tcp_syncookies) {
 			SCH_UNLOCK(sch);
 			if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 				log(LOG_DEBUG, "%s; %s: Spurious ACK, "
 				    "segment rejected (syncookies disabled)\n",
 				    s, __func__);
 			goto failed;
 		}
 		bzero(&scs, sizeof(scs));
 		sc = syncookie_lookup(inc, sch, &scs, th, to, *lsop);
 		SCH_UNLOCK(sch);
 		if (sc == NULL) {
 			if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 				log(LOG_DEBUG, "%s; %s: Segment failed "
 				    "SYNCOOKIE authentication, segment rejected "
 				    "(probably spoofed)\n", s, __func__);
 			goto failed;
 		}
 	} else {
 		/* Pull out the entry to unlock the bucket row. */
 		TAILQ_REMOVE(&sch->sch_bucket, sc, sc_hash);
 		sch->sch_length--;
 #ifdef TCP_OFFLOAD
 		if (ADDED_BY_TOE(sc)) {
 			struct toedev *tod = sc->sc_tod;
 
 			tod->tod_syncache_removed(tod, sc->sc_todctx);
 		}
 #endif
 		SCH_UNLOCK(sch);
 	}
 
 	/*
 	 * Segment validation:
 	 * ACK must match our initial sequence number + 1 (the SYN|ACK).
 	 */
 	if (th->th_ack != sc->sc_iss + 1) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: ACK %u != ISS+1 %u, segment "
 			    "rejected\n", s, __func__, th->th_ack, sc->sc_iss);
 		goto failed;
 	}
 
 	/*
 	 * The SEQ must fall in the window starting at the received
 	 * initial receive sequence number + 1 (the SYN).
 	 */
 	if (SEQ_LEQ(th->th_seq, sc->sc_irs) ||
 	    SEQ_GT(th->th_seq, sc->sc_irs + sc->sc_wnd)) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: SEQ %u != IRS+1 %u, segment "
 			    "rejected\n", s, __func__, th->th_seq, sc->sc_irs);
 		goto failed;
 	}
 
 	/*
 	 * If timestamps were not negotiated during SYN/ACK they
 	 * must not appear on any segment during this session.
 	 */
 	if (!(sc->sc_flags & SCF_TIMESTAMP) && (to->to_flags & TOF_TS)) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: Timestamp not expected, "
 			    "segment rejected\n", s, __func__);
 		goto failed;
 	}
 
 	/*
 	 * If timestamps were negotiated during SYN/ACK they should
 	 * appear on every segment during this session.
 	 * XXXAO: This is only informal as there have been unverified
 	 * reports of non-compliants stacks.
 	 */
 	if ((sc->sc_flags & SCF_TIMESTAMP) && !(to->to_flags & TOF_TS)) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL))) {
 			log(LOG_DEBUG, "%s; %s: Timestamp missing, "
 			    "no action\n", s, __func__);
 			free(s, M_TCPLOG);
 			s = NULL;
 		}
 	}
 
 	/*
 	 * If timestamps were negotiated the reflected timestamp
 	 * must be equal to what we actually sent in the SYN|ACK.
 	 */
 	if ((to->to_flags & TOF_TS) && to->to_tsecr != sc->sc_ts) {
 		if ((s = tcp_log_addrs(inc, th, NULL, NULL)))
 			log(LOG_DEBUG, "%s; %s: TSECR %u != TS %u, "
 			    "segment rejected\n",
 			    s, __func__, to->to_tsecr, sc->sc_ts);
 		goto failed;
 	}
 
 	*lsop = syncache_socket(sc, *lsop, m);
 
 	if (*lsop == NULL)
 		TCPSTAT_INC(tcps_sc_aborted);
 	else
 		TCPSTAT_INC(tcps_sc_completed);
 
 /* how do we find the inp for the new socket? */
 	if (sc != &scs)
 		syncache_free(sc);
 	return (1);
 failed:
 	if (sc != NULL && sc != &scs)
 		syncache_free(sc);
 	if (s != NULL)
 		free(s, M_TCPLOG);
 	*lsop = NULL;
 	return (0);
 }
 
 /*
  * Given a LISTEN socket and an inbound SYN request, add
  * this to the syn cache, and send back a segment:
  *	<SEQ=ISS><ACK=RCV_NXT><CTL=SYN,ACK>
  * to the source.
  *
  * IMPORTANT NOTE: We do _NOT_ ACK data that might accompany the SYN.
  * Doing so would require that we hold onto the data and deliver it
  * to the application.  However, if we are the target of a SYN-flood
  * DoS attack, an attacker could send data which would eventually
  * consume all available buffer space if it were ACKed.  By not ACKing
  * the data, we avoid this DoS scenario.
  */
 void
 syncache_add(struct in_conninfo *inc, struct tcpopt *to, struct tcphdr *th,
     struct inpcb *inp, struct socket **lsop, struct mbuf *m, void *tod,
     void *todctx)
 {
 	struct tcpcb *tp;
 	struct socket *so;
 	struct syncache *sc = NULL;
 	struct syncache_head *sch;
 	struct mbuf *ipopts = NULL;
 	u_int ltflags;
 	int win, sb_hiwat, ip_ttl, ip_tos;
 	char *s;
 #ifdef INET6
 	int autoflowlabel = 0;
 #endif
 #ifdef MAC
 	struct label *maclabel;
 #endif
 	struct syncache scs;
 	struct ucred *cred;
 
 	INP_WLOCK_ASSERT(inp);			/* listen socket */
 	KASSERT((th->th_flags & (TH_RST|TH_ACK|TH_SYN)) == TH_SYN,
 	    ("%s: unexpected tcp flags", __func__));
 
 	/*
 	 * Combine all so/tp operations very early to drop the INP lock as
 	 * soon as possible.
 	 */
 	so = *lsop;
 	tp = sototcpcb(so);
 	cred = crhold(so->so_cred);
 
 #ifdef INET6
 	if ((inc->inc_flags & INC_ISIPV6) &&
 	    (inp->inp_flags & IN6P_AUTOFLOWLABEL))
 		autoflowlabel = 1;
 #endif
 	ip_ttl = inp->inp_ip_ttl;
 	ip_tos = inp->inp_ip_tos;
 	win = sbspace(&so->so_rcv);
 	sb_hiwat = so->so_rcv.sb_hiwat;
 	ltflags = (tp->t_flags & (TF_NOOPT | TF_SIGNATURE));
 
 	/* By the time we drop the lock these should no longer be used. */
 	so = NULL;
 	tp = NULL;
 
 #ifdef MAC
 	if (mac_syncache_init(&maclabel) != 0) {
 		INP_WUNLOCK(inp);
 		goto done;
 	} else
 		mac_syncache_create(maclabel, inp);
 #endif
 	INP_WUNLOCK(inp);
 
 	/*
 	 * Remember the IP options, if any.
 	 */
 #ifdef INET6
 	if (!(inc->inc_flags & INC_ISIPV6))
 #endif
 #ifdef INET
 		ipopts = (m) ? ip_srcroute(m) : NULL;
 #else
 		ipopts = NULL;
 #endif
 
 	/*
 	 * See if we already have an entry for this connection.
 	 * If we do, resend the SYN,ACK, and reset the retransmit timer.
 	 *
 	 * XXX: should the syncache be re-initialized with the contents
 	 * of the new SYN here (which may have different options?)
 	 *
 	 * XXX: We do not check the sequence number to see if this is a
 	 * real retransmit or a new connection attempt.  The question is
 	 * how to handle such a case; either ignore it as spoofed, or
 	 * drop the current entry and create a new one?
 	 */
 	sc = syncache_lookup(inc, &sch);	/* returns locked entry */
 	SCH_LOCK_ASSERT(sch);
 	if (sc != NULL) {
 		TCPSTAT_INC(tcps_sc_dupsyn);
 		if (ipopts) {
 			/*
 			 * If we were remembering a previous source route,
 			 * forget it and use the new one we've been given.
 			 */
 			if (sc->sc_ipopts)
 				(void) m_free(sc->sc_ipopts);
 			sc->sc_ipopts = ipopts;
 		}
 		/*
 		 * Update timestamp if present.
 		 */
 		if ((sc->sc_flags & SCF_TIMESTAMP) && (to->to_flags & TOF_TS))
 			sc->sc_tsreflect = to->to_tsval;
 		else
 			sc->sc_flags &= ~SCF_TIMESTAMP;
 #ifdef MAC
 		/*
 		 * Since we have already unconditionally allocated label
 		 * storage, free it up.  The syncache entry will already
 		 * have an initialized label we can use.
 		 */
 		mac_syncache_destroy(&maclabel);
 #endif
 		/* Retransmit SYN|ACK and reset retransmit count. */
 		if ((s = tcp_log_addrs(&sc->sc_inc, th, NULL, NULL))) {
 			log(LOG_DEBUG, "%s; %s: Received duplicate SYN, "
 			    "resetting timer and retransmitting SYN|ACK\n",
 			    s, __func__);
 			free(s, M_TCPLOG);
 		}
 		if (syncache_respond(sc, sch, 1) == 0) {
 			sc->sc_rxmits = 0;
 			syncache_timeout(sc, sch, 1);
 			TCPSTAT_INC(tcps_sndacks);
 			TCPSTAT_INC(tcps_sndtotal);
 		}
 		SCH_UNLOCK(sch);
 		goto done;
 	}
 
 	sc = uma_zalloc(V_tcp_syncache.zone, M_NOWAIT | M_ZERO);
 	if (sc == NULL) {
 		/*
 		 * The zone allocator couldn't provide more entries.
 		 * Treat this as if the cache was full; drop the oldest
 		 * entry and insert the new one.
 		 */
 		TCPSTAT_INC(tcps_sc_zonefail);
 		if ((sc = TAILQ_LAST(&sch->sch_bucket, sch_head)) != NULL)
 			syncache_drop(sc, sch);
 		sc = uma_zalloc(V_tcp_syncache.zone, M_NOWAIT | M_ZERO);
 		if (sc == NULL) {
 			if (V_tcp_syncookies) {
 				bzero(&scs, sizeof(scs));
 				sc = &scs;
 			} else {
 				SCH_UNLOCK(sch);
 				if (ipopts)
 					(void) m_free(ipopts);
 				goto done;
 			}
 		}
 	}
 	
 	/*
 	 * Fill in the syncache values.
 	 */
 #ifdef MAC
 	sc->sc_label = maclabel;
 #endif
 	sc->sc_cred = cred;
 	cred = NULL;
 	sc->sc_ipopts = ipopts;
 	bcopy(inc, &sc->sc_inc, sizeof(struct in_conninfo));
 #ifdef INET6
 	if (!(inc->inc_flags & INC_ISIPV6))
 #endif
 	{
 		sc->sc_ip_tos = ip_tos;
 		sc->sc_ip_ttl = ip_ttl;
 	}
 #ifdef TCP_OFFLOAD
 	sc->sc_tod = tod;
 	sc->sc_todctx = todctx;
 #endif
 	sc->sc_irs = th->th_seq;
 	sc->sc_iss = arc4random();
 	sc->sc_flags = 0;
 	sc->sc_flowlabel = 0;
 
 	/*
 	 * Initial receive window: clip sbspace to [0 .. TCP_MAXWIN].
 	 * win was derived from socket earlier in the function.
 	 */
 	win = imax(win, 0);
 	win = imin(win, TCP_MAXWIN);
 	sc->sc_wnd = win;
 
 	if (V_tcp_do_rfc1323) {
 		/*
 		 * A timestamp received in a SYN makes
 		 * it ok to send timestamp requests and replies.
 		 */
 		if (to->to_flags & TOF_TS) {
 			sc->sc_tsreflect = to->to_tsval;
 			sc->sc_ts = tcp_ts_getticks();
 			sc->sc_flags |= SCF_TIMESTAMP;
 		}
 		if (to->to_flags & TOF_SCALE) {
 			int wscale = 0;
 
 			/*
 			 * Pick the smallest possible scaling factor that
 			 * will still allow us to scale up to sb_max, aka
 			 * kern.ipc.maxsockbuf.
 			 *
 			 * We do this because there are broken firewalls that
 			 * will corrupt the window scale option, leading to
 			 * the other endpoint believing that our advertised
 			 * window is unscaled.  At scale factors larger than
 			 * 5 the unscaled window will drop below 1500 bytes,
 			 * leading to serious problems when traversing these
 			 * broken firewalls.
 			 *
 			 * With the default maxsockbuf of 256K, a scale factor
 			 * of 3 will be chosen by this algorithm.  Those who
 			 * choose a larger maxsockbuf should watch out
 			 * for the compatiblity problems mentioned above.
 			 *
 			 * RFC1323: The Window field in a SYN (i.e., a <SYN>
 			 * or <SYN,ACK>) segment itself is never scaled.
 			 */
 			while (wscale < TCP_MAX_WINSHIFT &&
 			    (TCP_MAXWIN << wscale) < sb_max)
 				wscale++;
 			sc->sc_requested_r_scale = wscale;
 			sc->sc_requested_s_scale = to->to_wscale;
 			sc->sc_flags |= SCF_WINSCALE;
 		}
 	}
 #ifdef TCP_SIGNATURE
 	/*
 	 * If listening socket requested TCP digests, OR received SYN
 	 * contains the option, flag this in the syncache so that
 	 * syncache_respond() will do the right thing with the SYN+ACK.
 	 */
 	if (to->to_flags & TOF_SIGNATURE || ltflags & TF_SIGNATURE)
 		sc->sc_flags |= SCF_SIGNATURE;
 #endif
 	if (to->to_flags & TOF_SACKPERM)
 		sc->sc_flags |= SCF_SACK;
 	if (to->to_flags & TOF_MSS)
 		sc->sc_peer_mss = to->to_mss;	/* peer mss may be zero */
 	if (ltflags & TF_NOOPT)
 		sc->sc_flags |= SCF_NOOPT;
 	if ((th->th_flags & (TH_ECE|TH_CWR)) && V_tcp_do_ecn)
 		sc->sc_flags |= SCF_ECN;
 
 	if (V_tcp_syncookies)
 		sc->sc_iss = syncookie_generate(sch, sc);
 #ifdef INET6
 	if (autoflowlabel) {
 		if (V_tcp_syncookies)
 			sc->sc_flowlabel = sc->sc_iss;
 		else
 			sc->sc_flowlabel = ip6_randomflowlabel();
 		sc->sc_flowlabel = htonl(sc->sc_flowlabel) & IPV6_FLOWLABEL_MASK;
 	}
 #endif
 	SCH_UNLOCK(sch);
 
 	/*
 	 * Do a standard 3-way handshake.
 	 */
 	if (syncache_respond(sc, sch, 0) == 0) {
 		if (V_tcp_syncookies && V_tcp_syncookiesonly && sc != &scs)
 			syncache_free(sc);
 		else if (sc != &scs)
 			syncache_insert(sc, sch);   /* locks and unlocks sch */
 		TCPSTAT_INC(tcps_sndacks);
 		TCPSTAT_INC(tcps_sndtotal);
 	} else {
 		if (sc != &scs)
 			syncache_free(sc);
 		TCPSTAT_INC(tcps_sc_dropped);
 	}
 
 done:
 	if (cred != NULL)
 		crfree(cred);
 #ifdef MAC
 	if (sc == &scs)
 		mac_syncache_destroy(&maclabel);
 #endif
 	if (m) {
 		
 		*lsop = NULL;
 		m_freem(m);
 	}
 }
 
 static int
 syncache_respond(struct syncache *sc, struct syncache_head *sch, int locked)
 {
 	struct ip *ip = NULL;
 	struct mbuf *m;
 	struct tcphdr *th = NULL;
 	int optlen, error = 0;	/* Make compiler happy */
 	u_int16_t hlen, tlen, mssopt;
 	struct tcpopt to;
 #ifdef INET6
 	struct ip6_hdr *ip6 = NULL;
 #endif
 #ifdef TCP_SIGNATURE
 	struct secasvar *sav;
 #endif
 
 	hlen =
 #ifdef INET6
 	       (sc->sc_inc.inc_flags & INC_ISIPV6) ? sizeof(struct ip6_hdr) :
 #endif
 		sizeof(struct ip);
 	tlen = hlen + sizeof(struct tcphdr);
 
 	/* Determine MSS we advertize to other end of connection. */
 	mssopt = tcp_mssopt(&sc->sc_inc);
 	if (sc->sc_peer_mss)
 		mssopt = max( min(sc->sc_peer_mss, mssopt), V_tcp_minmss);
 
 	/* XXX: Assume that the entire packet will fit in a header mbuf. */
 	KASSERT(max_linkhdr + tlen + TCP_MAXOLEN <= MHLEN,
 	    ("syncache: mbuf too small"));
 
 	/* Create the IP+TCP header from scratch. */
 	m = m_gethdr(M_NOWAIT, MT_DATA);
 	if (m == NULL)
 		return (ENOBUFS);
 #ifdef MAC
 	mac_syncache_create_mbuf(sc->sc_label, m);
 #endif
 	m->m_data += max_linkhdr;
 	m->m_len = tlen;
 	m->m_pkthdr.len = tlen;
 	m->m_pkthdr.rcvif = NULL;
 
 #ifdef INET6
 	if (sc->sc_inc.inc_flags & INC_ISIPV6) {
 		ip6 = mtod(m, struct ip6_hdr *);
 		ip6->ip6_vfc = IPV6_VERSION;
 		ip6->ip6_nxt = IPPROTO_TCP;
 		ip6->ip6_src = sc->sc_inc.inc6_laddr;
 		ip6->ip6_dst = sc->sc_inc.inc6_faddr;
 		ip6->ip6_plen = htons(tlen - hlen);
 		/* ip6_hlim is set after checksum */
 		ip6->ip6_flow &= ~IPV6_FLOWLABEL_MASK;
 		ip6->ip6_flow |= sc->sc_flowlabel;
 
 		th = (struct tcphdr *)(ip6 + 1);
 	}
 #endif
 #if defined(INET6) && defined(INET)
 	else
 #endif
 #ifdef INET
 	{
 		ip = mtod(m, struct ip *);
 		ip->ip_v = IPVERSION;
 		ip->ip_hl = sizeof(struct ip) >> 2;
 		ip->ip_len = htons(tlen);
 		ip->ip_id = 0;
 		ip->ip_off = 0;
 		ip->ip_sum = 0;
 		ip->ip_p = IPPROTO_TCP;
 		ip->ip_src = sc->sc_inc.inc_laddr;
 		ip->ip_dst = sc->sc_inc.inc_faddr;
 		ip->ip_ttl = sc->sc_ip_ttl;
 		ip->ip_tos = sc->sc_ip_tos;
 
 		/*
 		 * See if we should do MTU discovery.  Route lookups are
 		 * expensive, so we will only unset the DF bit if:
 		 *
 		 *	1) path_mtu_discovery is disabled
 		 *	2) the SCF_UNREACH flag has been set
 		 */
 		if (V_path_mtu_discovery && ((sc->sc_flags & SCF_UNREACH) == 0))
 		       ip->ip_off |= htons(IP_DF);
 
 		th = (struct tcphdr *)(ip + 1);
 	}
 #endif /* INET */
 	th->th_sport = sc->sc_inc.inc_lport;
 	th->th_dport = sc->sc_inc.inc_fport;
 
 	th->th_seq = htonl(sc->sc_iss);
 	th->th_ack = htonl(sc->sc_irs + 1);
 	th->th_off = sizeof(struct tcphdr) >> 2;
 	th->th_x2 = 0;
 	th->th_flags = TH_SYN|TH_ACK;
 	th->th_win = htons(sc->sc_wnd);
 	th->th_urp = 0;
 
 	if (sc->sc_flags & SCF_ECN) {
 		th->th_flags |= TH_ECE;
 		TCPSTAT_INC(tcps_ecn_shs);
 	}
 
 	/* Tack on the TCP options. */
 	if ((sc->sc_flags & SCF_NOOPT) == 0) {
 		to.to_flags = 0;
 
 		to.to_mss = mssopt;
 		to.to_flags = TOF_MSS;
 		if (sc->sc_flags & SCF_WINSCALE) {
 			to.to_wscale = sc->sc_requested_r_scale;
 			to.to_flags |= TOF_SCALE;
 		}
 		if (sc->sc_flags & SCF_TIMESTAMP) {
 			/* Virgin timestamp or TCP cookie enhanced one. */
 			to.to_tsval = sc->sc_ts;
 			to.to_tsecr = sc->sc_tsreflect;
 			to.to_flags |= TOF_TS;
 		}
 		if (sc->sc_flags & SCF_SACK)
 			to.to_flags |= TOF_SACKPERM;
 #ifdef TCP_SIGNATURE
 		sav = NULL;
 		if (sc->sc_flags & SCF_SIGNATURE) {
 			sav = tcp_get_sav(m, IPSEC_DIR_OUTBOUND);
 			if (sav != NULL)
 				to.to_flags |= TOF_SIGNATURE;
 			else {
 
 				/*
 				 * We've got SCF_SIGNATURE flag
 				 * inherited from listening socket,
 				 * but no SADB key for given source
 				 * address. Assume signature is not
 				 * required and remove signature flag
 				 * instead of silently dropping
 				 * connection.
 				 */
 				if (locked == 0)
 					SCH_LOCK(sch);
 				sc->sc_flags &= ~SCF_SIGNATURE;
 				if (locked == 0)
 					SCH_UNLOCK(sch);
 			}
 		}
 #endif
 		optlen = tcp_addoptions(&to, (u_char *)(th + 1));
 
 		/* Adjust headers by option size. */
 		th->th_off = (sizeof(struct tcphdr) + optlen) >> 2;
 		m->m_len += optlen;
 		m->m_pkthdr.len += optlen;
 
 #ifdef TCP_SIGNATURE
 		if (sc->sc_flags & SCF_SIGNATURE)
 			tcp_signature_do_compute(m, 0, optlen,
 			    to.to_signature, sav);
 #endif
 #ifdef INET6
 		if (sc->sc_inc.inc_flags & INC_ISIPV6)
 			ip6->ip6_plen = htons(ntohs(ip6->ip6_plen) + optlen);
 		else
 #endif
 			ip->ip_len = htons(ntohs(ip->ip_len) + optlen);
 	} else
 		optlen = 0;
 
 	M_SETFIB(m, sc->sc_inc.inc_fibnum);
 	m->m_pkthdr.csum_data = offsetof(struct tcphdr, th_sum);
 #ifdef INET6
 	if (sc->sc_inc.inc_flags & INC_ISIPV6) {
 		m->m_pkthdr.csum_flags = CSUM_TCP_IPV6;
 		th->th_sum = in6_cksum_pseudo(ip6, tlen + optlen - hlen,
 		    IPPROTO_TCP, 0);
 		ip6->ip6_hlim = in6_selecthlim(NULL, NULL);
 #ifdef TCP_OFFLOAD
 		if (ADDED_BY_TOE(sc)) {
 			struct toedev *tod = sc->sc_tod;
 
 			error = tod->tod_syncache_respond(tod, sc->sc_todctx, m);
 
 			return (error);
 		}
 #endif
 		error = ip6_output(m, NULL, NULL, 0, NULL, NULL, NULL);
 	}
 #endif
 #if defined(INET6) && defined(INET)
 	else
 #endif
 #ifdef INET
 	{
 		m->m_pkthdr.csum_flags = CSUM_TCP;
 		th->th_sum = in_pseudo(ip->ip_src.s_addr, ip->ip_dst.s_addr,
 		    htons(tlen + optlen - hlen + IPPROTO_TCP));
 #ifdef TCP_OFFLOAD
 		if (ADDED_BY_TOE(sc)) {
 			struct toedev *tod = sc->sc_tod;
 
 			error = tod->tod_syncache_respond(tod, sc->sc_todctx, m);
 
 			return (error);
 		}
 #endif
 		error = ip_output(m, sc->sc_ipopts, NULL, 0, NULL, NULL);
 	}
 #endif
 	return (error);
 }
 
 /*
  * The purpose of syncookies is to handle spoofed SYN flooding DoS attacks
  * that exceed the capacity of the syncache by avoiding the storage of any
  * of the SYNs we receive.  Syncookies defend against blind SYN flooding
  * attacks where the attacker does not have access to our responses.
  *
  * Syncookies encode and include all necessary information about the
  * connection setup within the SYN|ACK that we send back.  That way we
  * can avoid keeping any local state until the ACK to our SYN|ACK returns
  * (if ever).  Normally the syncache and syncookies are running in parallel
  * with the latter taking over when the former is exhausted.  When matching
  * syncache entry is found the syncookie is ignored.
  *
  * The only reliable information persisting the 3WHS is our inital sequence
  * number ISS of 32 bits.  Syncookies embed a cryptographically sufficient
  * strong hash (MAC) value and a few bits of TCP SYN options in the ISS
  * of our SYN|ACK.  The MAC can be recomputed when the ACK to our SYN|ACK
  * returns and signifies a legitimate connection if it matches the ACK.
  *
  * The available space of 32 bits to store the hash and to encode the SYN
  * option information is very tight and we should have at least 24 bits for
  * the MAC to keep the number of guesses by blind spoofing reasonably high.
  *
  * SYN option information we have to encode to fully restore a connection:
  * MSS: is imporant to chose an optimal segment size to avoid IP level
  *   fragmentation along the path.  The common MSS values can be encoded
  *   in a 3-bit table.  Uncommon values are captured by the next lower value
  *   in the table leading to a slight increase in packetization overhead.
  * WSCALE: is necessary to allow large windows to be used for high delay-
  *   bandwidth product links.  Not scaling the window when it was initially
  *   negotiated is bad for performance as lack of scaling further decreases
  *   the apparent available send window.  We only need to encode the WSCALE
  *   we received from the remote end.  Our end can be recalculated at any
  *   time.  The common WSCALE values can be encoded in a 3-bit table.
  *   Uncommon values are captured by the next lower value in the table
  *   making us under-estimate the available window size halving our
  *   theoretically possible maximum throughput for that connection.
  * SACK: Greatly assists in packet loss recovery and requires 1 bit.
  * TIMESTAMP and SIGNATURE is not encoded because they are permanent options
  *   that are included in all segments on a connection.  We enable them when
  *   the ACK has them.
  *
  * Security of syncookies and attack vectors:
  *
  * The MAC is computed over (faddr||laddr||fport||lport||irs||flags||secmod)
  * together with the gloabl secret to make it unique per connection attempt.
  * Thus any change of any of those parameters results in a different MAC output
  * in an unpredictable way unless a collision is encountered.  24 bits of the
  * MAC are embedded into the ISS.
  *
  * To prevent replay attacks two rotating global secrets are updated with a
  * new random value every 15 seconds.  The life-time of a syncookie is thus
  * 15-30 seconds.
  *
  * Vector 1: Attacking the secret.  This requires finding a weakness in the
  * MAC itself or the way it is used here.  The attacker can do a chosen plain
  * text attack by varying and testing the all parameters under his control.
  * The strength depends on the size and randomness of the secret, and the
  * cryptographic security of the MAC function.  Due to the constant updating
  * of the secret the attacker has at most 29.999 seconds to find the secret
  * and launch spoofed connections.  After that he has to start all over again.
  *
  * Vector 2: Collision attack on the MAC of a single ACK.  With a 24 bit MAC
  * size an average of 4,823 attempts are required for a 50% chance of success
  * to spoof a single syncookie (birthday collision paradox).  However the
  * attacker is blind and doesn't know if one of his attempts succeeded unless
  * he has a side channel to interfere success from.  A single connection setup
  * success average of 90% requires 8,790 packets, 99.99% requires 17,578 packets.
  * This many attempts are required for each one blind spoofed connection.  For
  * every additional spoofed connection he has to launch another N attempts.
  * Thus for a sustained rate 100 spoofed connections per second approximately
  * 1,800,000 packets per second would have to be sent.
  *
  * NB: The MAC function should be fast so that it doesn't become a CPU
  * exhaustion attack vector itself.
  *
  * References:
  *  RFC4987 TCP SYN Flooding Attacks and Common Mitigations
  *  SYN cookies were first proposed by cryptographer Dan J. Bernstein in 1996
  *   http://cr.yp.to/syncookies.html    (overview)
  *   http://cr.yp.to/syncookies/archive (details)
  *
  *
  * Schematic construction of a syncookie enabled Initial Sequence Number:
  *  0        1         2         3
  *  12345678901234567890123456789012
  * |xxxxxxxxxxxxxxxxxxxxxxxxWWWMMMSP|
  *
  *  x 24 MAC (truncated)
  *  W  3 Send Window Scale index
  *  M  3 MSS index
  *  S  1 SACK permitted
  *  P  1 Odd/even secret
  */
 
 /*
  * Distribution and probability of certain MSS values.  Those in between are
  * rounded down to the next lower one.
  * [An Analysis of TCP Maximum Segment Sizes, S. Alcock and R. Nelson, 2011]
  *                            .2%  .3%   5%    7%    7%    20%   15%   45%
  */
 static int tcp_sc_msstab[] = { 216, 536, 1200, 1360, 1400, 1440, 1452, 1460 };
 
 /*
  * Distribution and probability of certain WSCALE values.  We have to map the
  * (send) window scale (shift) option with a range of 0-14 from 4 bits into 3
  * bits based on prevalence of certain values.  Where we don't have an exact
  * match for are rounded down to the next lower one letting us under-estimate
  * the true available window.  At the moment this would happen only for the
  * very uncommon values 3, 5 and those above 8 (more than 16MB socket buffer
  * and window size).  The absence of the WSCALE option (no scaling in either
  * direction) is encoded with index zero.
  * [WSCALE values histograms, Allman, 2012]
  *                            X 10 10 35  5  6 14 10%   by host
  *                            X 11  4  5  5 18 49  3%   by connections
  */
 static int tcp_sc_wstab[] = { 0, 0, 1, 2, 4, 6, 7, 8 };
 
 /*
  * Compute the MAC for the SYN cookie.  SIPHASH-2-4 is chosen for its speed
  * and good cryptographic properties.
  */
 static uint32_t
 syncookie_mac(struct in_conninfo *inc, tcp_seq irs, uint8_t flags,
     uint8_t *secbits, uintptr_t secmod)
 {
 	SIPHASH_CTX ctx;
 	uint32_t siphash[2];
 
 	SipHash24_Init(&ctx);
 	SipHash_SetKey(&ctx, secbits);
 	switch (inc->inc_flags & INC_ISIPV6) {
 #ifdef INET
 	case 0:
 		SipHash_Update(&ctx, &inc->inc_faddr, sizeof(inc->inc_faddr));
 		SipHash_Update(&ctx, &inc->inc_laddr, sizeof(inc->inc_laddr));
 		break;
 #endif
 #ifdef INET6
 	case INC_ISIPV6:
 		SipHash_Update(&ctx, &inc->inc6_faddr, sizeof(inc->inc6_faddr));
 		SipHash_Update(&ctx, &inc->inc6_laddr, sizeof(inc->inc6_laddr));
 		break;
 #endif
 	}
 	SipHash_Update(&ctx, &inc->inc_fport, sizeof(inc->inc_fport));
 	SipHash_Update(&ctx, &inc->inc_lport, sizeof(inc->inc_lport));
+	SipHash_Update(&ctx, &irs, sizeof(irs));
 	SipHash_Update(&ctx, &flags, sizeof(flags));
 	SipHash_Update(&ctx, &secmod, sizeof(secmod));
 	SipHash_Final((u_int8_t *)&siphash, &ctx);
 
 	return (siphash[0] ^ siphash[1]);
 }
 
 static tcp_seq
 syncookie_generate(struct syncache_head *sch, struct syncache *sc)
 {
 	u_int i, mss, secbit, wscale;
 	uint32_t iss, hash;
 	uint8_t *secbits;
 	union syncookie cookie;
 
 	SCH_LOCK_ASSERT(sch);
 
 	cookie.cookie = 0;
 
 	/* Map our computed MSS into the 3-bit index. */
 	mss = min(tcp_mssopt(&sc->sc_inc), max(sc->sc_peer_mss, V_tcp_minmss));
 	for (i = sizeof(tcp_sc_msstab) / sizeof(*tcp_sc_msstab) - 1;
 	     tcp_sc_msstab[i] > mss && i > 0;
 	     i--)
 		;
 	cookie.flags.mss_idx = i;
 
 	/*
 	 * Map the send window scale into the 3-bit index but only if
 	 * the wscale option was received.
 	 */
 	if (sc->sc_flags & SCF_WINSCALE) {
 		wscale = sc->sc_requested_s_scale;
 		for (i = sizeof(tcp_sc_wstab) / sizeof(*tcp_sc_wstab) - 1;
 		     tcp_sc_wstab[i] > wscale && i > 0;
 		     i--)
 			;
 		cookie.flags.wscale_idx = i;
 	}
 
 	/* Can we do SACK? */
 	if (sc->sc_flags & SCF_SACK)
 		cookie.flags.sack_ok = 1;
 
 	/* Which of the two secrets to use. */
 	secbit = sch->sch_sc->secret.oddeven & 0x1;
 	cookie.flags.odd_even = secbit;
 
 	secbits = sch->sch_sc->secret.key[secbit];
 	hash = syncookie_mac(&sc->sc_inc, sc->sc_irs, cookie.cookie, secbits,
 	    (uintptr_t)sch);
 
 	/*
 	 * Put the flags into the hash and XOR them to get better ISS number
 	 * variance.  This doesn't enhance the cryptographic strength and is
 	 * done to prevent the 8 cookie bits from showing up directly on the
 	 * wire.
 	 */
 	iss = hash & ~0xff;
 	iss |= cookie.cookie ^ (hash >> 24);
 
 	/* Randomize the timestamp. */
 	if (sc->sc_flags & SCF_TIMESTAMP) {
 		sc->sc_ts = arc4random();
 		sc->sc_tsoff = sc->sc_ts - tcp_ts_getticks();
 	}
 
 	TCPSTAT_INC(tcps_sc_sendcookie);
 	return (iss);
 }
 
 static struct syncache *
 syncookie_lookup(struct in_conninfo *inc, struct syncache_head *sch, 
     struct syncache *sc, struct tcphdr *th, struct tcpopt *to,
     struct socket *lso)
 {
 	uint32_t hash;
 	uint8_t *secbits;
 	tcp_seq ack, seq;
 	int wnd, wscale = 0;
 	union syncookie cookie;
 
 	SCH_LOCK_ASSERT(sch);
 
 	/*
 	 * Pull information out of SYN-ACK/ACK and revert sequence number
 	 * advances.
 	 */
 	ack = th->th_ack - 1;
 	seq = th->th_seq - 1;
 
 	/*
 	 * Unpack the flags containing enough information to restore the
 	 * connection.
 	 */
 	cookie.cookie = (ack & 0xff) ^ (ack >> 24);
 
 	/* Which of the two secrets to use. */
 	secbits = sch->sch_sc->secret.key[cookie.flags.odd_even];
 
 	hash = syncookie_mac(inc, seq, cookie.cookie, secbits, (uintptr_t)sch);
 
 	/* The recomputed hash matches the ACK if this was a genuine cookie. */
 	if ((ack & ~0xff) != (hash & ~0xff))
 		return (NULL);
 
 	/* Fill in the syncache values. */
 	sc->sc_flags = 0;
 	bcopy(inc, &sc->sc_inc, sizeof(struct in_conninfo));
 	sc->sc_ipopts = NULL;
 	
 	sc->sc_irs = seq;
 	sc->sc_iss = ack;
 
 	switch (inc->inc_flags & INC_ISIPV6) {
 #ifdef INET
 	case 0:
 		sc->sc_ip_ttl = sotoinpcb(lso)->inp_ip_ttl;
 		sc->sc_ip_tos = sotoinpcb(lso)->inp_ip_tos;
 		break;
 #endif
 #ifdef INET6
 	case INC_ISIPV6:
 		if (sotoinpcb(lso)->inp_flags & IN6P_AUTOFLOWLABEL)
 			sc->sc_flowlabel = sc->sc_iss & IPV6_FLOWLABEL_MASK;
 		break;
 #endif
 	}
 
 	sc->sc_peer_mss = tcp_sc_msstab[cookie.flags.mss_idx];
 
 	/* We can simply recompute receive window scale we sent earlier. */
 	while (wscale < TCP_MAX_WINSHIFT && (TCP_MAXWIN << wscale) < sb_max)
 		wscale++;
 
 	/* Only use wscale if it was enabled in the orignal SYN. */
 	if (cookie.flags.wscale_idx > 0) {
 		sc->sc_requested_r_scale = wscale;
 		sc->sc_requested_s_scale = tcp_sc_wstab[cookie.flags.wscale_idx];
 		sc->sc_flags |= SCF_WINSCALE;
 	}
 
 	wnd = sbspace(&lso->so_rcv);
 	wnd = imax(wnd, 0);
 	wnd = imin(wnd, TCP_MAXWIN);
 	sc->sc_wnd = wnd;
 
 	if (cookie.flags.sack_ok)
 		sc->sc_flags |= SCF_SACK;
 
 	if (to->to_flags & TOF_TS) {
 		sc->sc_flags |= SCF_TIMESTAMP;
 		sc->sc_tsreflect = to->to_tsval;
 		sc->sc_ts = to->to_tsecr;
 		sc->sc_tsoff = to->to_tsecr - tcp_ts_getticks();
 	}
 
 	if (to->to_flags & TOF_SIGNATURE)
 		sc->sc_flags |= SCF_SIGNATURE;
 
 	sc->sc_rxmits = 0;
 
 	TCPSTAT_INC(tcps_sc_recvcookie);
 	return (sc);
 }
 
 #ifdef INVARIANTS
 static int
 syncookie_cmp(struct in_conninfo *inc, struct syncache_head *sch,
     struct syncache *sc, struct tcphdr *th, struct tcpopt *to,
     struct socket *lso)
 {
 	struct syncache scs, *scx;
 	char *s;
 
 	bzero(&scs, sizeof(scs));
 	scx = syncookie_lookup(inc, sch, &scs, th, to, lso);
 
 	if ((s = tcp_log_addrs(inc, th, NULL, NULL)) == NULL)
 		return (0);
 
 	if (scx != NULL) {
 		if (sc->sc_peer_mss != scx->sc_peer_mss)
 			log(LOG_DEBUG, "%s; %s: mss different %i vs %i\n",
 			    s, __func__, sc->sc_peer_mss, scx->sc_peer_mss);
 
 		if (sc->sc_requested_r_scale != scx->sc_requested_r_scale)
 			log(LOG_DEBUG, "%s; %s: rwscale different %i vs %i\n",
 			    s, __func__, sc->sc_requested_r_scale,
 			    scx->sc_requested_r_scale);
 
 		if (sc->sc_requested_s_scale != scx->sc_requested_s_scale)
 			log(LOG_DEBUG, "%s; %s: swscale different %i vs %i\n",
 			    s, __func__, sc->sc_requested_s_scale,
 			    scx->sc_requested_s_scale);
 
 		if ((sc->sc_flags & SCF_SACK) != (scx->sc_flags & SCF_SACK))
 			log(LOG_DEBUG, "%s; %s: SACK different\n", s, __func__);
 	}
 
 	if (s != NULL)
 		free(s, M_TCPLOG);
 	return (0);
 }
 #endif /* INVARIANTS */
 
 static void
 syncookie_reseed(void *arg)
 {
 	struct tcp_syncache *sc = arg;
 	uint8_t *secbits;
 	int secbit;
 
 	/*
 	 * Reseeding the secret doesn't have to be protected by a lock.
 	 * It only must be ensured that the new random values are visible
 	 * to all CPUs in a SMP environment.  The atomic with release
 	 * semantics ensures that.
 	 */
 	secbit = (sc->secret.oddeven & 0x1) ? 0 : 1;
 	secbits = sc->secret.key[secbit];
 	arc4rand(secbits, SYNCOOKIE_SECRET_SIZE, 0);
 	atomic_add_rel_int(&sc->secret.oddeven, 1);
 
 	/* Reschedule ourself. */
 	callout_schedule(&sc->secret.reseed, SYNCOOKIE_LIFETIME * hz);
 }
 
 /*
  * Returns the current number of syncache entries.  This number
  * will probably change before you get around to calling 
  * syncache_pcblist.
  */
 int
 syncache_pcbcount(void)
 {
 	struct syncache_head *sch;
 	int count, i;
 
 	for (count = 0, i = 0; i < V_tcp_syncache.hashsize; i++) {
 		/* No need to lock for a read. */
 		sch = &V_tcp_syncache.hashbase[i];
 		count += sch->sch_length;
 	}
 	return count;
 }
 
 /*
  * Exports the syncache entries to userland so that netstat can display
  * them alongside the other sockets.  This function is intended to be
  * called only from tcp_pcblist.
  *
  * Due to concurrency on an active system, the number of pcbs exported
  * may have no relation to max_pcbs.  max_pcbs merely indicates the
  * amount of space the caller allocated for this function to use.
  */
 int
 syncache_pcblist(struct sysctl_req *req, int max_pcbs, int *pcbs_exported)
 {
 	struct xtcpcb xt;
 	struct syncache *sc;
 	struct syncache_head *sch;
 	int count, error, i;
 
 	for (count = 0, error = 0, i = 0; i < V_tcp_syncache.hashsize; i++) {
 		sch = &V_tcp_syncache.hashbase[i];
 		SCH_LOCK(sch);
 		TAILQ_FOREACH(sc, &sch->sch_bucket, sc_hash) {
 			if (count >= max_pcbs) {
 				SCH_UNLOCK(sch);
 				goto exit;
 			}
 			if (cr_cansee(req->td->td_ucred, sc->sc_cred) != 0)
 				continue;
 			bzero(&xt, sizeof(xt));
 			xt.xt_len = sizeof(xt);
 			if (sc->sc_inc.inc_flags & INC_ISIPV6)
 				xt.xt_inp.inp_vflag = INP_IPV6;
 			else
 				xt.xt_inp.inp_vflag = INP_IPV4;
 			bcopy(&sc->sc_inc, &xt.xt_inp.inp_inc, sizeof (struct in_conninfo));
 			xt.xt_tp.t_inpcb = &xt.xt_inp;
 			xt.xt_tp.t_state = TCPS_SYN_RECEIVED;
 			xt.xt_socket.xso_protocol = IPPROTO_TCP;
 			xt.xt_socket.xso_len = sizeof (struct xsocket);
 			xt.xt_socket.so_type = SOCK_STREAM;
 			xt.xt_socket.so_state = SS_ISCONNECTING;
 			error = SYSCTL_OUT(req, &xt, sizeof xt);
 			if (error) {
 				SCH_UNLOCK(sch);
 				goto exit;
 			}
 			count++;
 		}
 		SCH_UNLOCK(sch);
 	}
 exit:
 	*pcbs_exported = count;
 	return error;
 }
Index: projects/clang360-import/sys/ufs/ffs/ffs_softdep.c
===================================================================
--- projects/clang360-import/sys/ufs/ffs/ffs_softdep.c	(revision 277944)
+++ projects/clang360-import/sys/ufs/ffs/ffs_softdep.c	(revision 277945)
@@ -1,14137 +1,14146 @@
 /*-
  * Copyright 1998, 2000 Marshall Kirk McKusick.
  * Copyright 2009, 2010 Jeffrey W. Roberson <jeff@FreeBSD.org>
  * All rights reserved.
  *
  * The soft updates code is derived from the appendix of a University
  * of Michigan technical report (Gregory R. Ganger and Yale N. Patt,
  * "Soft Updates: A Solution to the Metadata Update Problem in File
  * Systems", CSE-TR-254-95, August 1995).
  *
  * Further information about soft updates can be obtained from:
  *
  *	Marshall Kirk McKusick		http://www.mckusick.com/softdep/
  *	1614 Oxford Street		mckusick@mckusick.com
  *	Berkeley, CA 94709-1608		+1-510-843-9542
  *	USA
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  *
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY THE AUTHORS ``AS IS'' AND ANY EXPRESS OR
  * IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES
  * OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE DISCLAIMED.
  * IN NO EVENT SHALL THE AUTHORS BE LIABLE FOR ANY DIRECT, INDIRECT,
  * INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING,
  * BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS
  * OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND
  * ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR
  * TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE
  * USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
  *
  *	from: @(#)ffs_softdep.c	9.59 (McKusick) 6/21/00
  */
 
 #include <sys/cdefs.h>
 __FBSDID("$FreeBSD$");
 
 #include "opt_ffs.h"
 #include "opt_quota.h"
 #include "opt_ddb.h"
 
 /*
  * For now we want the safety net that the DEBUG flag provides.
  */
 #ifndef DEBUG
 #define DEBUG
 #endif
 
 #include <sys/param.h>
 #include <sys/kernel.h>
 #include <sys/systm.h>
 #include <sys/bio.h>
 #include <sys/buf.h>
 #include <sys/kdb.h>
 #include <sys/kthread.h>
 #include <sys/ktr.h>
 #include <sys/limits.h>
 #include <sys/lock.h>
 #include <sys/malloc.h>
 #include <sys/mount.h>
 #include <sys/mutex.h>
 #include <sys/namei.h>
 #include <sys/priv.h>
 #include <sys/proc.h>
 #include <sys/rwlock.h>
 #include <sys/stat.h>
 #include <sys/sysctl.h>
 #include <sys/syslog.h>
 #include <sys/vnode.h>
 #include <sys/conf.h>
 
 #include <ufs/ufs/dir.h>
 #include <ufs/ufs/extattr.h>
 #include <ufs/ufs/quota.h>
 #include <ufs/ufs/inode.h>
 #include <ufs/ufs/ufsmount.h>
 #include <ufs/ffs/fs.h>
 #include <ufs/ffs/softdep.h>
 #include <ufs/ffs/ffs_extern.h>
 #include <ufs/ufs/ufs_extern.h>
 
 #include <vm/vm.h>
 #include <vm/vm_extern.h>
 #include <vm/vm_object.h>
 
 #include <geom/geom.h>
 
 #include <ddb/ddb.h>
 
 #define	KTR_SUJ	0	/* Define to KTR_SPARE. */
 
 #ifndef SOFTUPDATES
 
 int
 softdep_flushfiles(oldmnt, flags, td)
 	struct mount *oldmnt;
 	int flags;
 	struct thread *td;
 {
 
 	panic("softdep_flushfiles called");
 }
 
 int
 softdep_mount(devvp, mp, fs, cred)
 	struct vnode *devvp;
 	struct mount *mp;
 	struct fs *fs;
 	struct ucred *cred;
 {
 
 	return (0);
 }
 
 void
 softdep_initialize()
 {
 
 	return;
 }
 
 void
 softdep_uninitialize()
 {
 
 	return;
 }
 
 void
 softdep_unmount(mp)
 	struct mount *mp;
 {
 
 	panic("softdep_unmount called");
 }
 
 void
 softdep_setup_sbupdate(ump, fs, bp)
 	struct ufsmount *ump;
 	struct fs *fs;
 	struct buf *bp;
 {
 
 	panic("softdep_setup_sbupdate called");
 }
 
 void
 softdep_setup_inomapdep(bp, ip, newinum, mode)
 	struct buf *bp;
 	struct inode *ip;
 	ino_t newinum;
 	int mode;
 {
 
 	panic("softdep_setup_inomapdep called");
 }
 
 void
 softdep_setup_blkmapdep(bp, mp, newblkno, frags, oldfrags)
 	struct buf *bp;
 	struct mount *mp;
 	ufs2_daddr_t newblkno;
 	int frags;
 	int oldfrags;
 {
 
 	panic("softdep_setup_blkmapdep called");
 }
 
 void
 softdep_setup_allocdirect(ip, lbn, newblkno, oldblkno, newsize, oldsize, bp)
 	struct inode *ip;
 	ufs_lbn_t lbn;
 	ufs2_daddr_t newblkno;
 	ufs2_daddr_t oldblkno;
 	long newsize;
 	long oldsize;
 	struct buf *bp;
 {
 	
 	panic("softdep_setup_allocdirect called");
 }
 
 void
 softdep_setup_allocext(ip, lbn, newblkno, oldblkno, newsize, oldsize, bp)
 	struct inode *ip;
 	ufs_lbn_t lbn;
 	ufs2_daddr_t newblkno;
 	ufs2_daddr_t oldblkno;
 	long newsize;
 	long oldsize;
 	struct buf *bp;
 {
 	
 	panic("softdep_setup_allocext called");
 }
 
 void
 softdep_setup_allocindir_page(ip, lbn, bp, ptrno, newblkno, oldblkno, nbp)
 	struct inode *ip;
 	ufs_lbn_t lbn;
 	struct buf *bp;
 	int ptrno;
 	ufs2_daddr_t newblkno;
 	ufs2_daddr_t oldblkno;
 	struct buf *nbp;
 {
 
 	panic("softdep_setup_allocindir_page called");
 }
 
 void
 softdep_setup_allocindir_meta(nbp, ip, bp, ptrno, newblkno)
 	struct buf *nbp;
 	struct inode *ip;
 	struct buf *bp;
 	int ptrno;
 	ufs2_daddr_t newblkno;
 {
 
 	panic("softdep_setup_allocindir_meta called");
 }
 
 void
 softdep_journal_freeblocks(ip, cred, length, flags)
 	struct inode *ip;
 	struct ucred *cred;
 	off_t length;
 	int flags;
 {
 	
 	panic("softdep_journal_freeblocks called");
 }
 
 void
 softdep_journal_fsync(ip)
 	struct inode *ip;
 {
 
 	panic("softdep_journal_fsync called");
 }
 
 void
 softdep_setup_freeblocks(ip, length, flags)
 	struct inode *ip;
 	off_t length;
 	int flags;
 {
 	
 	panic("softdep_setup_freeblocks called");
 }
 
 void
 softdep_freefile(pvp, ino, mode)
 		struct vnode *pvp;
 		ino_t ino;
 		int mode;
 {
 
 	panic("softdep_freefile called");
 }
 
 int
 softdep_setup_directory_add(bp, dp, diroffset, newinum, newdirbp, isnewblk)
 	struct buf *bp;
 	struct inode *dp;
 	off_t diroffset;
 	ino_t newinum;
 	struct buf *newdirbp;
 	int isnewblk;
 {
 
 	panic("softdep_setup_directory_add called");
 }
 
 void
 softdep_change_directoryentry_offset(bp, dp, base, oldloc, newloc, entrysize)
 	struct buf *bp;
 	struct inode *dp;
 	caddr_t base;
 	caddr_t oldloc;
 	caddr_t newloc;
 	int entrysize;
 {
 
 	panic("softdep_change_directoryentry_offset called");
 }
 
 void
 softdep_setup_remove(bp, dp, ip, isrmdir)
 	struct buf *bp;
 	struct inode *dp;
 	struct inode *ip;
 	int isrmdir;
 {
 	
 	panic("softdep_setup_remove called");
 }
 
 void
 softdep_setup_directory_change(bp, dp, ip, newinum, isrmdir)
 	struct buf *bp;
 	struct inode *dp;
 	struct inode *ip;
 	ino_t newinum;
 	int isrmdir;
 {
 
 	panic("softdep_setup_directory_change called");
 }
 
 void
 softdep_setup_blkfree(mp, bp, blkno, frags, wkhd)
 	struct mount *mp;
 	struct buf *bp;
 	ufs2_daddr_t blkno;
 	int frags;
 	struct workhead *wkhd;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_inofree(mp, bp, ino, wkhd)
 	struct mount *mp;
 	struct buf *bp;
 	ino_t ino;
 	struct workhead *wkhd;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_unlink(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_link(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_revert_link(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_rmdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_revert_rmdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_create(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_revert_create(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_mkdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_revert_mkdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 void
 softdep_setup_dotdot_link(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 int
 softdep_prealloc(vp, waitok)
 	struct vnode *vp;
 	int waitok;
 {
 
 	panic("%s called", __FUNCTION__);
 }
 
 int
 softdep_journal_lookup(mp, vpp)
 	struct mount *mp;
 	struct vnode **vpp;
 {
 
 	return (ENOENT);
 }
 
 void
 softdep_change_linkcnt(ip)
 	struct inode *ip;
 {
 
 	panic("softdep_change_linkcnt called");
 }
 
 void 
 softdep_load_inodeblock(ip)
 	struct inode *ip;
 {
 
 	panic("softdep_load_inodeblock called");
 }
 
 void
 softdep_update_inodeblock(ip, bp, waitfor)
 	struct inode *ip;
 	struct buf *bp;
 	int waitfor;
 {
 
 	panic("softdep_update_inodeblock called");
 }
 
 int
 softdep_fsync(vp)
 	struct vnode *vp;	/* the "in_core" copy of the inode */
 {
 
 	return (0);
 }
 
 void
 softdep_fsync_mountdev(vp)
 	struct vnode *vp;
 {
 
 	return;
 }
 
 int
 softdep_flushworklist(oldmnt, countp, td)
 	struct mount *oldmnt;
 	int *countp;
 	struct thread *td;
 {
 
 	*countp = 0;
 	return (0);
 }
 
 int
 softdep_sync_metadata(struct vnode *vp)
 {
 
 	panic("softdep_sync_metadata called");
 }
 
 int
 softdep_sync_buf(struct vnode *vp, struct buf *bp, int waitfor)
 {
 
 	panic("softdep_sync_buf called");
 }
 
 int
 softdep_slowdown(vp)
 	struct vnode *vp;
 {
 
 	panic("softdep_slowdown called");
 }
 
 int
 softdep_request_cleanup(fs, vp, cred, resource)
 	struct fs *fs;
 	struct vnode *vp;
 	struct ucred *cred;
 	int resource;
 {
 
 	return (0);
 }
 
 int
 softdep_check_suspend(struct mount *mp,
 		      struct vnode *devvp,
 		      int softdep_depcnt,
 		      int softdep_accdepcnt,
 		      int secondary_writes,
 		      int secondary_accwrites)
 {
 	struct bufobj *bo;
 	int error;
 	
 	(void) softdep_depcnt,
 	(void) softdep_accdepcnt;
 
 	bo = &devvp->v_bufobj;
 	ASSERT_BO_WLOCKED(bo);
 
 	MNT_ILOCK(mp);
 	while (mp->mnt_secondary_writes != 0) {
 		BO_UNLOCK(bo);
 		msleep(&mp->mnt_secondary_writes, MNT_MTX(mp),
 		    (PUSER - 1) | PDROP, "secwr", 0);
 		BO_LOCK(bo);
 		MNT_ILOCK(mp);
 	}
 
 	/*
 	 * Reasons for needing more work before suspend:
 	 * - Dirty buffers on devvp.
 	 * - Secondary writes occurred after start of vnode sync loop
 	 */
 	error = 0;
 	if (bo->bo_numoutput > 0 ||
 	    bo->bo_dirty.bv_cnt > 0 ||
 	    secondary_writes != 0 ||
 	    mp->mnt_secondary_writes != 0 ||
 	    secondary_accwrites != mp->mnt_secondary_accwrites)
 		error = EAGAIN;
 	BO_UNLOCK(bo);
 	return (error);
 }
 
 void
 softdep_get_depcounts(struct mount *mp,
 		      int *softdepactivep,
 		      int *softdepactiveaccp)
 {
 	(void) mp;
 	*softdepactivep = 0;
 	*softdepactiveaccp = 0;
 }
 
 void
 softdep_buf_append(bp, wkhd)
 	struct buf *bp;
 	struct workhead *wkhd;
 {
 
 	panic("softdep_buf_appendwork called");
 }
 
 void
 softdep_inode_append(ip, cred, wkhd)
 	struct inode *ip;
 	struct ucred *cred;
 	struct workhead *wkhd;
 {
 
 	panic("softdep_inode_appendwork called");
 }
 
 void
 softdep_freework(wkhd)
 	struct workhead *wkhd;
 {
 
 	panic("softdep_freework called");
 }
 
 #else
 
 FEATURE(softupdates, "FFS soft-updates support");
 
 static SYSCTL_NODE(_debug, OID_AUTO, softdep, CTLFLAG_RW, 0,
     "soft updates stats");
 static SYSCTL_NODE(_debug_softdep, OID_AUTO, total, CTLFLAG_RW, 0,
     "total dependencies allocated");
 static SYSCTL_NODE(_debug_softdep, OID_AUTO, highuse, CTLFLAG_RW, 0,
     "high use dependencies allocated");
 static SYSCTL_NODE(_debug_softdep, OID_AUTO, current, CTLFLAG_RW, 0,
     "current dependencies allocated");
 static SYSCTL_NODE(_debug_softdep, OID_AUTO, write, CTLFLAG_RW, 0,
     "current dependencies written");
 
 unsigned long dep_current[D_LAST + 1];
 unsigned long dep_highuse[D_LAST + 1];
 unsigned long dep_total[D_LAST + 1];
 unsigned long dep_write[D_LAST + 1];
 
 #define	SOFTDEP_TYPE(type, str, long)					\
     static MALLOC_DEFINE(M_ ## type, #str, long);			\
     SYSCTL_ULONG(_debug_softdep_total, OID_AUTO, str, CTLFLAG_RD,	\
 	&dep_total[D_ ## type], 0, "");					\
     SYSCTL_ULONG(_debug_softdep_current, OID_AUTO, str, CTLFLAG_RD, 	\
 	&dep_current[D_ ## type], 0, "");				\
     SYSCTL_ULONG(_debug_softdep_highuse, OID_AUTO, str, CTLFLAG_RD, 	\
 	&dep_highuse[D_ ## type], 0, "");				\
     SYSCTL_ULONG(_debug_softdep_write, OID_AUTO, str, CTLFLAG_RD, 	\
 	&dep_write[D_ ## type], 0, "");
 
 SOFTDEP_TYPE(PAGEDEP, pagedep, "File page dependencies"); 
 SOFTDEP_TYPE(INODEDEP, inodedep, "Inode dependencies");
 SOFTDEP_TYPE(BMSAFEMAP, bmsafemap,
     "Block or frag allocated from cyl group map");
 SOFTDEP_TYPE(NEWBLK, newblk, "New block or frag allocation dependency");
 SOFTDEP_TYPE(ALLOCDIRECT, allocdirect, "Block or frag dependency for an inode");
 SOFTDEP_TYPE(INDIRDEP, indirdep, "Indirect block dependencies");
 SOFTDEP_TYPE(ALLOCINDIR, allocindir, "Block dependency for an indirect block");
 SOFTDEP_TYPE(FREEFRAG, freefrag, "Previously used frag for an inode");
 SOFTDEP_TYPE(FREEBLKS, freeblks, "Blocks freed from an inode");
 SOFTDEP_TYPE(FREEFILE, freefile, "Inode deallocated");
 SOFTDEP_TYPE(DIRADD, diradd, "New directory entry");
 SOFTDEP_TYPE(MKDIR, mkdir, "New directory");
 SOFTDEP_TYPE(DIRREM, dirrem, "Directory entry deleted");
 SOFTDEP_TYPE(NEWDIRBLK, newdirblk, "Unclaimed new directory block");
 SOFTDEP_TYPE(FREEWORK, freework, "free an inode block");
 SOFTDEP_TYPE(FREEDEP, freedep, "track a block free");
 SOFTDEP_TYPE(JADDREF, jaddref, "Journal inode ref add");
 SOFTDEP_TYPE(JREMREF, jremref, "Journal inode ref remove");
 SOFTDEP_TYPE(JMVREF, jmvref, "Journal inode ref move");
 SOFTDEP_TYPE(JNEWBLK, jnewblk, "Journal new block");
 SOFTDEP_TYPE(JFREEBLK, jfreeblk, "Journal free block");
 SOFTDEP_TYPE(JFREEFRAG, jfreefrag, "Journal free frag");
 SOFTDEP_TYPE(JSEG, jseg, "Journal segment");
 SOFTDEP_TYPE(JSEGDEP, jsegdep, "Journal segment complete");
 SOFTDEP_TYPE(SBDEP, sbdep, "Superblock write dependency");
 SOFTDEP_TYPE(JTRUNC, jtrunc, "Journal inode truncation");
 SOFTDEP_TYPE(JFSYNC, jfsync, "Journal fsync complete");
 
 static MALLOC_DEFINE(M_SENTINEL, "sentinel", "Worklist sentinel");
 
 static MALLOC_DEFINE(M_SAVEDINO, "savedino", "Saved inodes");
 static MALLOC_DEFINE(M_JBLOCKS, "jblocks", "Journal block locations");
 static MALLOC_DEFINE(M_MOUNTDATA, "softdep", "Softdep per-mount data");
 
 #define M_SOFTDEP_FLAGS	(M_WAITOK)
 
 /* 
  * translate from workitem type to memory type
  * MUST match the defines above, such that memtype[D_XXX] == M_XXX
  */
 static struct malloc_type *memtype[] = {
 	M_PAGEDEP,
 	M_INODEDEP,
 	M_BMSAFEMAP,
 	M_NEWBLK,
 	M_ALLOCDIRECT,
 	M_INDIRDEP,
 	M_ALLOCINDIR,
 	M_FREEFRAG,
 	M_FREEBLKS,
 	M_FREEFILE,
 	M_DIRADD,
 	M_MKDIR,
 	M_DIRREM,
 	M_NEWDIRBLK,
 	M_FREEWORK,
 	M_FREEDEP,
 	M_JADDREF,
 	M_JREMREF,
 	M_JMVREF,
 	M_JNEWBLK,
 	M_JFREEBLK,
 	M_JFREEFRAG,
 	M_JSEG,
 	M_JSEGDEP,
 	M_SBDEP,
 	M_JTRUNC,
 	M_JFSYNC,
 	M_SENTINEL
 };
 
 #define DtoM(type) (memtype[type])
 
 /*
  * Names of malloc types.
  */
 #define TYPENAME(type)  \
 	((unsigned)(type) <= D_LAST ? memtype[type]->ks_shortdesc : "???")
 /*
  * End system adaptation definitions.
  */
 
 #define	DOTDOT_OFFSET	offsetof(struct dirtemplate, dotdot_ino)
 #define	DOT_OFFSET	offsetof(struct dirtemplate, dot_ino)
 
 /*
  * Internal function prototypes.
  */
 static	void check_clear_deps(struct mount *);
 static	void softdep_error(char *, int);
 static	int softdep_process_worklist(struct mount *, int);
 static	int softdep_waitidle(struct mount *, int);
 static	void drain_output(struct vnode *);
 static	struct buf *getdirtybuf(struct buf *, struct rwlock *, int);
 static	void clear_remove(struct mount *);
 static	void clear_inodedeps(struct mount *);
 static	void unlinked_inodedep(struct mount *, struct inodedep *);
 static	void clear_unlinked_inodedep(struct inodedep *);
 static	struct inodedep *first_unlinked_inodedep(struct ufsmount *);
 static	int flush_pagedep_deps(struct vnode *, struct mount *,
 	    struct diraddhd *);
 static	int free_pagedep(struct pagedep *);
 static	int flush_newblk_dep(struct vnode *, struct mount *, ufs_lbn_t);
 static	int flush_inodedep_deps(struct vnode *, struct mount *, ino_t);
 static	int flush_deplist(struct allocdirectlst *, int, int *);
 static	int sync_cgs(struct mount *, int);
 static	int handle_written_filepage(struct pagedep *, struct buf *);
 static	int handle_written_sbdep(struct sbdep *, struct buf *);
 static	void initiate_write_sbdep(struct sbdep *);
 static	void diradd_inode_written(struct diradd *, struct inodedep *);
 static	int handle_written_indirdep(struct indirdep *, struct buf *,
 	    struct buf**);
 static	int handle_written_inodeblock(struct inodedep *, struct buf *);
 static	int jnewblk_rollforward(struct jnewblk *, struct fs *, struct cg *,
 	    uint8_t *);
 static	int handle_written_bmsafemap(struct bmsafemap *, struct buf *);
 static	void handle_written_jaddref(struct jaddref *);
 static	void handle_written_jremref(struct jremref *);
 static	void handle_written_jseg(struct jseg *, struct buf *);
 static	void handle_written_jnewblk(struct jnewblk *);
 static	void handle_written_jblkdep(struct jblkdep *);
 static	void handle_written_jfreefrag(struct jfreefrag *);
 static	void complete_jseg(struct jseg *);
 static	void complete_jsegs(struct jseg *);
 static	void jseg_write(struct ufsmount *ump, struct jseg *, uint8_t *);
 static	void jaddref_write(struct jaddref *, struct jseg *, uint8_t *);
 static	void jremref_write(struct jremref *, struct jseg *, uint8_t *);
 static	void jmvref_write(struct jmvref *, struct jseg *, uint8_t *);
 static	void jtrunc_write(struct jtrunc *, struct jseg *, uint8_t *);
 static	void jfsync_write(struct jfsync *, struct jseg *, uint8_t *data);
 static	void jnewblk_write(struct jnewblk *, struct jseg *, uint8_t *);
 static	void jfreeblk_write(struct jfreeblk *, struct jseg *, uint8_t *);
 static	void jfreefrag_write(struct jfreefrag *, struct jseg *, uint8_t *);
 static	inline void inoref_write(struct inoref *, struct jseg *,
 	    struct jrefrec *);
 static	void handle_allocdirect_partdone(struct allocdirect *,
 	    struct workhead *);
 static	struct jnewblk *cancel_newblk(struct newblk *, struct worklist *,
 	    struct workhead *);
 static	void indirdep_complete(struct indirdep *);
 static	int indirblk_lookup(struct mount *, ufs2_daddr_t);
 static	void indirblk_insert(struct freework *);
 static	void indirblk_remove(struct freework *);
 static	void handle_allocindir_partdone(struct allocindir *);
 static	void initiate_write_filepage(struct pagedep *, struct buf *);
 static	void initiate_write_indirdep(struct indirdep*, struct buf *);
 static	void handle_written_mkdir(struct mkdir *, int);
 static	int jnewblk_rollback(struct jnewblk *, struct fs *, struct cg *,
 	    uint8_t *);
 static	void initiate_write_bmsafemap(struct bmsafemap *, struct buf *);
 static	void initiate_write_inodeblock_ufs1(struct inodedep *, struct buf *);
 static	void initiate_write_inodeblock_ufs2(struct inodedep *, struct buf *);
 static	void handle_workitem_freefile(struct freefile *);
 static	int handle_workitem_remove(struct dirrem *, int);
 static	struct dirrem *newdirrem(struct buf *, struct inode *,
 	    struct inode *, int, struct dirrem **);
 static	struct indirdep *indirdep_lookup(struct mount *, struct inode *,
 	    struct buf *);
 static	void cancel_indirdep(struct indirdep *, struct buf *,
 	    struct freeblks *);
 static	void free_indirdep(struct indirdep *);
 static	void free_diradd(struct diradd *, struct workhead *);
 static	void merge_diradd(struct inodedep *, struct diradd *);
 static	void complete_diradd(struct diradd *);
 static	struct diradd *diradd_lookup(struct pagedep *, int);
 static	struct jremref *cancel_diradd_dotdot(struct inode *, struct dirrem *,
 	    struct jremref *);
 static	struct jremref *cancel_mkdir_dotdot(struct inode *, struct dirrem *,
 	    struct jremref *);
 static	void cancel_diradd(struct diradd *, struct dirrem *, struct jremref *,
 	    struct jremref *, struct jremref *);
 static	void dirrem_journal(struct dirrem *, struct jremref *, struct jremref *,
 	    struct jremref *);
 static	void cancel_allocindir(struct allocindir *, struct buf *bp,
 	    struct freeblks *, int);
 static	int setup_trunc_indir(struct freeblks *, struct inode *,
 	    ufs_lbn_t, ufs_lbn_t, ufs2_daddr_t);
 static	void complete_trunc_indir(struct freework *);
 static	void trunc_indirdep(struct indirdep *, struct freeblks *, struct buf *,
 	    int);
 static	void complete_mkdir(struct mkdir *);
 static	void free_newdirblk(struct newdirblk *);
 static	void free_jremref(struct jremref *);
 static	void free_jaddref(struct jaddref *);
 static	void free_jsegdep(struct jsegdep *);
 static	void free_jsegs(struct jblocks *);
 static	void rele_jseg(struct jseg *);
 static	void free_jseg(struct jseg *, struct jblocks *);
 static	void free_jnewblk(struct jnewblk *);
 static	void free_jblkdep(struct jblkdep *);
 static	void free_jfreefrag(struct jfreefrag *);
 static	void free_freedep(struct freedep *);
 static	void journal_jremref(struct dirrem *, struct jremref *,
 	    struct inodedep *);
 static	void cancel_jnewblk(struct jnewblk *, struct workhead *);
 static	int cancel_jaddref(struct jaddref *, struct inodedep *,
 	    struct workhead *);
 static	void cancel_jfreefrag(struct jfreefrag *);
 static	inline void setup_freedirect(struct freeblks *, struct inode *,
 	    int, int);
 static	inline void setup_freeext(struct freeblks *, struct inode *, int, int);
 static	inline void setup_freeindir(struct freeblks *, struct inode *, int,
 	    ufs_lbn_t, int);
 static	inline struct freeblks *newfreeblks(struct mount *, struct inode *);
 static	void freeblks_free(struct ufsmount *, struct freeblks *, int);
 static	void indir_trunc(struct freework *, ufs2_daddr_t, ufs_lbn_t);
 static	ufs2_daddr_t blkcount(struct fs *, ufs2_daddr_t, off_t);
 static	int trunc_check_buf(struct buf *, int *, ufs_lbn_t, int, int);
 static	void trunc_dependencies(struct inode *, struct freeblks *, ufs_lbn_t,
 	    int, int);
 static	void trunc_pages(struct inode *, off_t, ufs2_daddr_t, int);
 static 	int cancel_pagedep(struct pagedep *, struct freeblks *, int);
 static	int deallocate_dependencies(struct buf *, struct freeblks *, int);
 static	void newblk_freefrag(struct newblk*);
 static	void free_newblk(struct newblk *);
 static	void cancel_allocdirect(struct allocdirectlst *,
 	    struct allocdirect *, struct freeblks *);
 static	int check_inode_unwritten(struct inodedep *);
 static	int free_inodedep(struct inodedep *);
 static	void freework_freeblock(struct freework *);
 static	void freework_enqueue(struct freework *);
 static	int handle_workitem_freeblocks(struct freeblks *, int);
 static	int handle_complete_freeblocks(struct freeblks *, int);
 static	void handle_workitem_indirblk(struct freework *);
 static	void handle_written_freework(struct freework *);
 static	void merge_inode_lists(struct allocdirectlst *,struct allocdirectlst *);
 static	struct worklist *jnewblk_merge(struct worklist *, struct worklist *,
 	    struct workhead *);
 static	struct freefrag *setup_allocindir_phase2(struct buf *, struct inode *,
 	    struct inodedep *, struct allocindir *, ufs_lbn_t);
 static	struct allocindir *newallocindir(struct inode *, int, ufs2_daddr_t,
 	    ufs2_daddr_t, ufs_lbn_t);
 static	void handle_workitem_freefrag(struct freefrag *);
 static	struct freefrag *newfreefrag(struct inode *, ufs2_daddr_t, long,
 	    ufs_lbn_t);
 static	void allocdirect_merge(struct allocdirectlst *,
 	    struct allocdirect *, struct allocdirect *);
 static	struct freefrag *allocindir_merge(struct allocindir *,
 	    struct allocindir *);
 static	int bmsafemap_find(struct bmsafemap_hashhead *, int,
 	    struct bmsafemap **);
 static	struct bmsafemap *bmsafemap_lookup(struct mount *, struct buf *,
 	    int cg, struct bmsafemap *);
 static	int newblk_find(struct newblk_hashhead *, ufs2_daddr_t, int,
 	    struct newblk **);
 static	int newblk_lookup(struct mount *, ufs2_daddr_t, int, struct newblk **);
 static	int inodedep_find(struct inodedep_hashhead *, ino_t,
 	    struct inodedep **);
 static	int inodedep_lookup(struct mount *, ino_t, int, struct inodedep **);
 static	int pagedep_lookup(struct mount *, struct buf *bp, ino_t, ufs_lbn_t,
 	    int, struct pagedep **);
 static	int pagedep_find(struct pagedep_hashhead *, ino_t, ufs_lbn_t,
 	    struct pagedep **);
 static	void pause_timer(void *);
 static	int request_cleanup(struct mount *, int);
 static	int process_worklist_item(struct mount *, int, int);
 static	void process_removes(struct vnode *);
 static	void process_truncates(struct vnode *);
 static	void jwork_move(struct workhead *, struct workhead *);
 static	void jwork_insert(struct workhead *, struct jsegdep *);
 static	void add_to_worklist(struct worklist *, int);
 static	void wake_worklist(struct worklist *);
 static	void wait_worklist(struct worklist *, char *);
 static	void remove_from_worklist(struct worklist *);
 static	void softdep_flush(void *);
 static	void softdep_flushjournal(struct mount *);
 static	int softdep_speedup(struct ufsmount *);
 static	void worklist_speedup(struct mount *);
 static	int journal_mount(struct mount *, struct fs *, struct ucred *);
 static	void journal_unmount(struct ufsmount *);
 static	int journal_space(struct ufsmount *, int);
 static	void journal_suspend(struct ufsmount *);
 static	int journal_unsuspend(struct ufsmount *ump);
 static	void softdep_prelink(struct vnode *, struct vnode *);
 static	void add_to_journal(struct worklist *);
 static	void remove_from_journal(struct worklist *);
 static	void softdep_process_journal(struct mount *, struct worklist *, int);
 static	struct jremref *newjremref(struct dirrem *, struct inode *,
 	    struct inode *ip, off_t, nlink_t);
 static	struct jaddref *newjaddref(struct inode *, ino_t, off_t, int16_t,
 	    uint16_t);
 static	inline void newinoref(struct inoref *, ino_t, ino_t, off_t, nlink_t,
 	    uint16_t);
 static	inline struct jsegdep *inoref_jseg(struct inoref *);
 static	struct jmvref *newjmvref(struct inode *, ino_t, off_t, off_t);
 static	struct jfreeblk *newjfreeblk(struct freeblks *, ufs_lbn_t,
 	    ufs2_daddr_t, int);
 static	void adjust_newfreework(struct freeblks *, int);
 static	struct jtrunc *newjtrunc(struct freeblks *, off_t, int);
 static	void move_newblock_dep(struct jaddref *, struct inodedep *);
 static	void cancel_jfreeblk(struct freeblks *, ufs2_daddr_t);
 static	struct jfreefrag *newjfreefrag(struct freefrag *, struct inode *,
 	    ufs2_daddr_t, long, ufs_lbn_t);
 static	struct freework *newfreework(struct ufsmount *, struct freeblks *,
 	    struct freework *, ufs_lbn_t, ufs2_daddr_t, int, int, int);
 static	int jwait(struct worklist *, int);
 static	struct inodedep *inodedep_lookup_ip(struct inode *);
 static	int bmsafemap_backgroundwrite(struct bmsafemap *, struct buf *);
 static	struct freefile *handle_bufwait(struct inodedep *, struct workhead *);
 static	void handle_jwork(struct workhead *);
 static	struct mkdir *setup_newdir(struct diradd *, ino_t, ino_t, struct buf *,
 	    struct mkdir **);
 static	struct jblocks *jblocks_create(void);
 static	ufs2_daddr_t jblocks_alloc(struct jblocks *, int, int *);
 static	void jblocks_free(struct jblocks *, struct mount *, int);
 static	void jblocks_destroy(struct jblocks *);
 static	void jblocks_add(struct jblocks *, ufs2_daddr_t, int);
 
 /*
  * Exported softdep operations.
  */
 static	void softdep_disk_io_initiation(struct buf *);
 static	void softdep_disk_write_complete(struct buf *);
 static	void softdep_deallocate_dependencies(struct buf *);
 static	int softdep_count_dependencies(struct buf *bp, int);
 
 /*
  * Global lock over all of soft updates.
  */
 static struct mtx lk;
 MTX_SYSINIT(softdep_lock, &lk, "Global Softdep Lock", MTX_DEF);
 
 #define ACQUIRE_GBLLOCK(lk)	mtx_lock(lk)
 #define FREE_GBLLOCK(lk)	mtx_unlock(lk)
 #define GBLLOCK_OWNED(lk)	mtx_assert((lk), MA_OWNED)
 
 /*
  * Per-filesystem soft-updates locking.
  */
 #define LOCK_PTR(ump)		(&(ump)->um_softdep->sd_fslock)
 #define TRY_ACQUIRE_LOCK(ump)	rw_try_wlock(&(ump)->um_softdep->sd_fslock)
 #define ACQUIRE_LOCK(ump)	rw_wlock(&(ump)->um_softdep->sd_fslock)
 #define FREE_LOCK(ump)		rw_wunlock(&(ump)->um_softdep->sd_fslock)
 #define LOCK_OWNED(ump)		rw_assert(&(ump)->um_softdep->sd_fslock, \
 				    RA_WLOCKED)
 
 #define	BUF_AREC(bp)		lockallowrecurse(&(bp)->b_lock)
 #define	BUF_NOREC(bp)		lockdisablerecurse(&(bp)->b_lock)
 
 /*
  * Worklist queue management.
  * These routines require that the lock be held.
  */
 #ifndef /* NOT */ DEBUG
 #define WORKLIST_INSERT(head, item) do {	\
 	(item)->wk_state |= ONWORKLIST;		\
 	LIST_INSERT_HEAD(head, item, wk_list);	\
 } while (0)
 #define WORKLIST_REMOVE(item) do {		\
 	(item)->wk_state &= ~ONWORKLIST;	\
 	LIST_REMOVE(item, wk_list);		\
 } while (0)
 #define WORKLIST_INSERT_UNLOCKED	WORKLIST_INSERT
 #define WORKLIST_REMOVE_UNLOCKED	WORKLIST_REMOVE
 
 #else /* DEBUG */
 static	void worklist_insert(struct workhead *, struct worklist *, int);
 static	void worklist_remove(struct worklist *, int);
 
 #define WORKLIST_INSERT(head, item) worklist_insert(head, item, 1)
 #define WORKLIST_INSERT_UNLOCKED(head, item) worklist_insert(head, item, 0)
 #define WORKLIST_REMOVE(item) worklist_remove(item, 1)
 #define WORKLIST_REMOVE_UNLOCKED(item) worklist_remove(item, 0)
 
 static void
 worklist_insert(head, item, locked)
 	struct workhead *head;
 	struct worklist *item;
 	int locked;
 {
 
 	if (locked)
 		LOCK_OWNED(VFSTOUFS(item->wk_mp));
 	if (item->wk_state & ONWORKLIST)
 		panic("worklist_insert: %p %s(0x%X) already on list",
 		    item, TYPENAME(item->wk_type), item->wk_state);
 	item->wk_state |= ONWORKLIST;
 	LIST_INSERT_HEAD(head, item, wk_list);
 }
 
 static void
 worklist_remove(item, locked)
 	struct worklist *item;
 	int locked;
 {
 
 	if (locked)
 		LOCK_OWNED(VFSTOUFS(item->wk_mp));
 	if ((item->wk_state & ONWORKLIST) == 0)
 		panic("worklist_remove: %p %s(0x%X) not on list",
 		    item, TYPENAME(item->wk_type), item->wk_state);
 	item->wk_state &= ~ONWORKLIST;
 	LIST_REMOVE(item, wk_list);
 }
 #endif /* DEBUG */
 
 /*
  * Merge two jsegdeps keeping only the oldest one as newer references
  * can't be discarded until after older references.
  */
 static inline struct jsegdep *
 jsegdep_merge(struct jsegdep *one, struct jsegdep *two)
 {
 	struct jsegdep *swp;
 
 	if (two == NULL)
 		return (one);
 
 	if (one->jd_seg->js_seq > two->jd_seg->js_seq) {
 		swp = one;
 		one = two;
 		two = swp;
 	}
 	WORKLIST_REMOVE(&two->jd_list);
 	free_jsegdep(two);
 
 	return (one);
 }
 
 /*
  * If two freedeps are compatible free one to reduce list size.
  */
 static inline struct freedep *
 freedep_merge(struct freedep *one, struct freedep *two)
 {
 	if (two == NULL)
 		return (one);
 
 	if (one->fd_freework == two->fd_freework) {
 		WORKLIST_REMOVE(&two->fd_list);
 		free_freedep(two);
 	}
 	return (one);
 }
 
 /*
  * Move journal work from one list to another.  Duplicate freedeps and
  * jsegdeps are coalesced to keep the lists as small as possible.
  */
 static void
 jwork_move(dst, src)
 	struct workhead *dst;
 	struct workhead *src;
 {
 	struct freedep *freedep;
 	struct jsegdep *jsegdep;
 	struct worklist *wkn;
 	struct worklist *wk;
 
 	KASSERT(dst != src,
 	    ("jwork_move: dst == src"));
 	freedep = NULL;
 	jsegdep = NULL;
 	LIST_FOREACH_SAFE(wk, dst, wk_list, wkn) {
 		if (wk->wk_type == D_JSEGDEP)
 			jsegdep = jsegdep_merge(WK_JSEGDEP(wk), jsegdep);
 		else if (wk->wk_type == D_FREEDEP)
 			freedep = freedep_merge(WK_FREEDEP(wk), freedep);
 	}
 
 	while ((wk = LIST_FIRST(src)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		WORKLIST_INSERT(dst, wk);
 		if (wk->wk_type == D_JSEGDEP) {
 			jsegdep = jsegdep_merge(WK_JSEGDEP(wk), jsegdep);
 			continue;
 		}
 		if (wk->wk_type == D_FREEDEP)
 			freedep = freedep_merge(WK_FREEDEP(wk), freedep);
 	}
 }
 
 static void
 jwork_insert(dst, jsegdep)
 	struct workhead *dst;
 	struct jsegdep *jsegdep;
 {
 	struct jsegdep *jsegdepn;
 	struct worklist *wk;
 
 	LIST_FOREACH(wk, dst, wk_list)
 		if (wk->wk_type == D_JSEGDEP)
 			break;
 	if (wk == NULL) {
 		WORKLIST_INSERT(dst, &jsegdep->jd_list);
 		return;
 	}
 	jsegdepn = WK_JSEGDEP(wk);
 	if (jsegdep->jd_seg->js_seq < jsegdepn->jd_seg->js_seq) {
 		WORKLIST_REMOVE(wk);
 		free_jsegdep(jsegdepn);
 		WORKLIST_INSERT(dst, &jsegdep->jd_list);
 	} else
 		free_jsegdep(jsegdep);
 }
 
 /*
  * Routines for tracking and managing workitems.
  */
 static	void workitem_free(struct worklist *, int);
 static	void workitem_alloc(struct worklist *, int, struct mount *);
 static	void workitem_reassign(struct worklist *, int);
 
 #define	WORKITEM_FREE(item, type) \
 	workitem_free((struct worklist *)(item), (type))
 #define	WORKITEM_REASSIGN(item, type) \
 	workitem_reassign((struct worklist *)(item), (type))
 
 static void
 workitem_free(item, type)
 	struct worklist *item;
 	int type;
 {
 	struct ufsmount *ump;
 
 #ifdef DEBUG
 	if (item->wk_state & ONWORKLIST)
 		panic("workitem_free: %s(0x%X) still on list",
 		    TYPENAME(item->wk_type), item->wk_state);
 	if (item->wk_type != type && type != D_NEWBLK)
 		panic("workitem_free: type mismatch %s != %s",
 		    TYPENAME(item->wk_type), TYPENAME(type));
 #endif
 	if (item->wk_state & IOWAITING)
 		wakeup(item);
 	ump = VFSTOUFS(item->wk_mp);
 	LOCK_OWNED(ump);
 	KASSERT(ump->softdep_deps > 0,
 	    ("workitem_free: %s: softdep_deps going negative",
 	    ump->um_fs->fs_fsmnt));
 	if (--ump->softdep_deps == 0 && ump->softdep_req)
 		wakeup(&ump->softdep_deps);
 	KASSERT(dep_current[item->wk_type] > 0,
 	    ("workitem_free: %s: dep_current[%s] going negative",
 	    ump->um_fs->fs_fsmnt, TYPENAME(item->wk_type)));
 	KASSERT(ump->softdep_curdeps[item->wk_type] > 0,
 	    ("workitem_free: %s: softdep_curdeps[%s] going negative",
 	    ump->um_fs->fs_fsmnt, TYPENAME(item->wk_type)));
 	atomic_subtract_long(&dep_current[item->wk_type], 1);
 	ump->softdep_curdeps[item->wk_type] -= 1;
 	free(item, DtoM(type));
 }
 
 static void
 workitem_alloc(item, type, mp)
 	struct worklist *item;
 	int type;
 	struct mount *mp;
 {
 	struct ufsmount *ump;
 
 	item->wk_type = type;
 	item->wk_mp = mp;
 	item->wk_state = 0;
 
 	ump = VFSTOUFS(mp);
 	ACQUIRE_GBLLOCK(&lk);
 	dep_current[type]++;
 	if (dep_current[type] > dep_highuse[type])
 		dep_highuse[type] = dep_current[type];
 	dep_total[type]++;
 	FREE_GBLLOCK(&lk);
 	ACQUIRE_LOCK(ump);
 	ump->softdep_curdeps[type] += 1;
 	ump->softdep_deps++;
 	ump->softdep_accdeps++;
 	FREE_LOCK(ump);
 }
 
 static void
 workitem_reassign(item, newtype)
 	struct worklist *item;
 	int newtype;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(item->wk_mp);
 	LOCK_OWNED(ump);
 	KASSERT(ump->softdep_curdeps[item->wk_type] > 0,
 	    ("workitem_reassign: %s: softdep_curdeps[%s] going negative",
 	    VFSTOUFS(item->wk_mp)->um_fs->fs_fsmnt, TYPENAME(item->wk_type)));
 	ump->softdep_curdeps[item->wk_type] -= 1;
 	ump->softdep_curdeps[newtype] += 1;
 	KASSERT(dep_current[item->wk_type] > 0,
 	    ("workitem_reassign: %s: dep_current[%s] going negative",
 	    VFSTOUFS(item->wk_mp)->um_fs->fs_fsmnt, TYPENAME(item->wk_type)));
 	ACQUIRE_GBLLOCK(&lk);
 	dep_current[newtype]++;
 	dep_current[item->wk_type]--;
 	if (dep_current[newtype] > dep_highuse[newtype])
 		dep_highuse[newtype] = dep_current[newtype];
 	dep_total[newtype]++;
 	FREE_GBLLOCK(&lk);
 	item->wk_type = newtype;
 }
 
 /*
  * Workitem queue management
  */
 static int max_softdeps;	/* maximum number of structs before slowdown */
 static int tickdelay = 2;	/* number of ticks to pause during slowdown */
 static int proc_waiting;	/* tracks whether we have a timeout posted */
 static int *stat_countp;	/* statistic to count in proc_waiting timeout */
 static struct callout softdep_callout;
 static int req_clear_inodedeps;	/* syncer process flush some inodedeps */
 static int req_clear_remove;	/* syncer process flush some freeblks */
 static int softdep_flushcache = 0; /* Should we do BIO_FLUSH? */
 
 /*
  * runtime statistics
  */
 static int stat_flush_threads;	/* number of softdep flushing threads */
 static int stat_worklist_push;	/* number of worklist cleanups */
 static int stat_blk_limit_push;	/* number of times block limit neared */
 static int stat_ino_limit_push;	/* number of times inode limit neared */
 static int stat_blk_limit_hit;	/* number of times block slowdown imposed */
 static int stat_ino_limit_hit;	/* number of times inode slowdown imposed */
 static int stat_sync_limit_hit;	/* number of synchronous slowdowns imposed */
 static int stat_indir_blk_ptrs;	/* bufs redirtied as indir ptrs not written */
 static int stat_inode_bitmap;	/* bufs redirtied as inode bitmap not written */
 static int stat_direct_blk_ptrs;/* bufs redirtied as direct ptrs not written */
 static int stat_dir_entry;	/* bufs redirtied as dir entry cannot write */
 static int stat_jaddref;	/* bufs redirtied as ino bitmap can not write */
 static int stat_jnewblk;	/* bufs redirtied as blk bitmap can not write */
 static int stat_journal_min;	/* Times hit journal min threshold */
 static int stat_journal_low;	/* Times hit journal low threshold */
 static int stat_journal_wait;	/* Times blocked in jwait(). */
 static int stat_jwait_filepage;	/* Times blocked in jwait() for filepage. */
 static int stat_jwait_freeblks;	/* Times blocked in jwait() for freeblks. */
 static int stat_jwait_inode;	/* Times blocked in jwait() for inodes. */
 static int stat_jwait_newblk;	/* Times blocked in jwait() for newblks. */
 static int stat_cleanup_high_delay; /* Maximum cleanup delay (in ticks) */
 static int stat_cleanup_blkrequests; /* Number of block cleanup requests */
 static int stat_cleanup_inorequests; /* Number of inode cleanup requests */
 static int stat_cleanup_retries; /* Number of cleanups that needed to flush */
 static int stat_cleanup_failures; /* Number of cleanup requests that failed */
 static int stat_emptyjblocks; /* Number of potentially empty journal blocks */
 
 SYSCTL_INT(_debug_softdep, OID_AUTO, max_softdeps, CTLFLAG_RW,
     &max_softdeps, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, tickdelay, CTLFLAG_RW,
     &tickdelay, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, flush_threads, CTLFLAG_RD,
     &stat_flush_threads, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, worklist_push, CTLFLAG_RW,
     &stat_worklist_push, 0,"");
 SYSCTL_INT(_debug_softdep, OID_AUTO, blk_limit_push, CTLFLAG_RW,
     &stat_blk_limit_push, 0,"");
 SYSCTL_INT(_debug_softdep, OID_AUTO, ino_limit_push, CTLFLAG_RW,
     &stat_ino_limit_push, 0,"");
 SYSCTL_INT(_debug_softdep, OID_AUTO, blk_limit_hit, CTLFLAG_RW,
     &stat_blk_limit_hit, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, ino_limit_hit, CTLFLAG_RW,
     &stat_ino_limit_hit, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, sync_limit_hit, CTLFLAG_RW,
     &stat_sync_limit_hit, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, indir_blk_ptrs, CTLFLAG_RW,
     &stat_indir_blk_ptrs, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, inode_bitmap, CTLFLAG_RW,
     &stat_inode_bitmap, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, direct_blk_ptrs, CTLFLAG_RW,
     &stat_direct_blk_ptrs, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, dir_entry, CTLFLAG_RW,
     &stat_dir_entry, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, jaddref_rollback, CTLFLAG_RW,
     &stat_jaddref, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, jnewblk_rollback, CTLFLAG_RW,
     &stat_jnewblk, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, journal_low, CTLFLAG_RW,
     &stat_journal_low, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, journal_min, CTLFLAG_RW,
     &stat_journal_min, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, journal_wait, CTLFLAG_RW,
     &stat_journal_wait, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, jwait_filepage, CTLFLAG_RW,
     &stat_jwait_filepage, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, jwait_freeblks, CTLFLAG_RW,
     &stat_jwait_freeblks, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, jwait_inode, CTLFLAG_RW,
     &stat_jwait_inode, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, jwait_newblk, CTLFLAG_RW,
     &stat_jwait_newblk, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, cleanup_blkrequests, CTLFLAG_RW,
     &stat_cleanup_blkrequests, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, cleanup_inorequests, CTLFLAG_RW,
     &stat_cleanup_inorequests, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, cleanup_high_delay, CTLFLAG_RW,
     &stat_cleanup_high_delay, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, cleanup_retries, CTLFLAG_RW,
     &stat_cleanup_retries, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, cleanup_failures, CTLFLAG_RW,
     &stat_cleanup_failures, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, flushcache, CTLFLAG_RW,
     &softdep_flushcache, 0, "");
 SYSCTL_INT(_debug_softdep, OID_AUTO, emptyjblocks, CTLFLAG_RD,
     &stat_emptyjblocks, 0, "");
 
 SYSCTL_DECL(_vfs_ffs);
 
 /* Whether to recompute the summary at mount time */
 static int compute_summary_at_mount = 0;
 SYSCTL_INT(_vfs_ffs, OID_AUTO, compute_summary_at_mount, CTLFLAG_RW,
 	   &compute_summary_at_mount, 0, "Recompute summary at mount");
 static int print_threads = 0;
 SYSCTL_INT(_debug_softdep, OID_AUTO, print_threads, CTLFLAG_RW,
     &print_threads, 0, "Notify flusher thread start/stop");
 
 /* List of all filesystems mounted with soft updates */
 static TAILQ_HEAD(, mount_softdeps) softdepmounts;
 
 /*
  * This function cleans the worklist for a filesystem.
  * Each filesystem running with soft dependencies gets its own
  * thread to run in this function. The thread is started up in
  * softdep_mount and shutdown in softdep_unmount. They show up
  * as part of the kernel "bufdaemon" process whose process
  * entry is available in bufdaemonproc.
  */
 static int searchfailed;
 extern struct proc *bufdaemonproc;
 static void
 softdep_flush(addr)
 	void *addr;
 {
 	struct mount *mp;
 	struct thread *td;
 	struct ufsmount *ump;
 
 	td = curthread;
 	td->td_pflags |= TDP_NORUNNINGBUF;
 	mp = (struct mount *)addr;
 	ump = VFSTOUFS(mp);
 	atomic_add_int(&stat_flush_threads, 1);
+	ACQUIRE_LOCK(ump);
+	ump->softdep_flags &= ~FLUSH_STARTING;
+	wakeup(&ump->softdep_flushtd);
+	FREE_LOCK(ump);
 	if (print_threads) {
 		if (stat_flush_threads == 1)
 			printf("Running %s at pid %d\n", bufdaemonproc->p_comm,
 			    bufdaemonproc->p_pid);
 		printf("Start thread %s\n", td->td_name);
 	}
 	for (;;) {	
 		while (softdep_process_worklist(mp, 0) > 0 ||
 		    (MOUNTEDSUJ(mp) &&
 		    VFSTOUFS(mp)->softdep_jblocks->jb_suspended))
 			kthread_suspend_check();
 		ACQUIRE_LOCK(ump);
-		if ((ump->softdep_flags & FLUSH_CLEANUP) == 0)
+		while ((ump->softdep_flags & (FLUSH_CLEANUP | FLUSH_EXIT)) == 0)
 			msleep(&ump->softdep_flushtd, LOCK_PTR(ump), PVM,
 			    "sdflush", hz / 2);
 		ump->softdep_flags &= ~FLUSH_CLEANUP;
 		/*
 		 * Check to see if we are done and need to exit.
 		 */
 		if ((ump->softdep_flags & FLUSH_EXIT) == 0) {
 			FREE_LOCK(ump);
 			continue;
 		}
 		ump->softdep_flags &= ~FLUSH_EXIT;
 		FREE_LOCK(ump);
 		wakeup(&ump->softdep_flags);
 		if (print_threads)
 			printf("Stop thread %s: searchfailed %d, did cleanups %d\n", td->td_name, searchfailed, ump->um_softdep->sd_cleanups);
 		atomic_subtract_int(&stat_flush_threads, 1);
 		kthread_exit();
 		panic("kthread_exit failed\n");
 	}
 }
 
 static void
 worklist_speedup(mp)
 	struct mount *mp;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	if ((ump->softdep_flags & (FLUSH_CLEANUP | FLUSH_EXIT)) == 0) {
 		ump->softdep_flags |= FLUSH_CLEANUP;
-		if (ump->softdep_flushtd->td_wchan == &ump->softdep_flushtd)
-			wakeup(&ump->softdep_flushtd);
+		wakeup(&ump->softdep_flushtd);
 	}
 }
 
 static int
 softdep_speedup(ump)
 	struct ufsmount *ump;
 {
 	struct ufsmount *altump;
 	struct mount_softdeps *sdp;
 
 	LOCK_OWNED(ump);
 	worklist_speedup(ump->um_mountp);
 	bd_speedup();
 	/*
 	 * If we have global shortages, then we need other
 	 * filesystems to help with the cleanup. Here we wakeup a
 	 * flusher thread for a filesystem that is over its fair
 	 * share of resources.
 	 */
 	if (req_clear_inodedeps || req_clear_remove) {
 		ACQUIRE_GBLLOCK(&lk);
 		TAILQ_FOREACH(sdp, &softdepmounts, sd_next) {
 			if ((altump = sdp->sd_ump) == ump)
 				continue;
 			if (((req_clear_inodedeps &&
 			    altump->softdep_curdeps[D_INODEDEP] >
 			    max_softdeps / stat_flush_threads) ||
 			    (req_clear_remove &&
 			    altump->softdep_curdeps[D_DIRREM] >
 			    (max_softdeps / 2) / stat_flush_threads)) &&
 			    TRY_ACQUIRE_LOCK(altump))
 				break;
 		}
 		if (sdp == NULL) {
 			searchfailed++;
 			FREE_GBLLOCK(&lk);
 		} else {
 			/*
 			 * Move to the end of the list so we pick a
 			 * different one on out next try.
 			 */
 			TAILQ_REMOVE(&softdepmounts, sdp, sd_next);
 			TAILQ_INSERT_TAIL(&softdepmounts, sdp, sd_next);
 			FREE_GBLLOCK(&lk);
 			if ((altump->softdep_flags &
 			    (FLUSH_CLEANUP | FLUSH_EXIT)) == 0) {
 				altump->softdep_flags |= FLUSH_CLEANUP;
 				altump->um_softdep->sd_cleanups++;
-				if (altump->softdep_flushtd->td_wchan ==
-				    &altump->softdep_flushtd) {
-					wakeup(&altump->softdep_flushtd);
-				}
+				wakeup(&altump->softdep_flushtd);
 			}
 			FREE_LOCK(altump);
 		}
 	}
 	return (speedup_syncer());
 }
 
 /*
  * Add an item to the end of the work queue.
  * This routine requires that the lock be held.
  * This is the only routine that adds items to the list.
  * The following routine is the only one that removes items
  * and does so in order from first to last.
  */
 
 #define	WK_HEAD		0x0001	/* Add to HEAD. */
 #define	WK_NODELAY	0x0002	/* Process immediately. */
 
 static void
 add_to_worklist(wk, flags)
 	struct worklist *wk;
 	int flags;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(wk->wk_mp);
 	LOCK_OWNED(ump);
 	if (wk->wk_state & ONWORKLIST)
 		panic("add_to_worklist: %s(0x%X) already on list",
 		    TYPENAME(wk->wk_type), wk->wk_state);
 	wk->wk_state |= ONWORKLIST;
 	if (ump->softdep_on_worklist == 0) {
 		LIST_INSERT_HEAD(&ump->softdep_workitem_pending, wk, wk_list);
 		ump->softdep_worklist_tail = wk;
 	} else if (flags & WK_HEAD) {
 		LIST_INSERT_HEAD(&ump->softdep_workitem_pending, wk, wk_list);
 	} else {
 		LIST_INSERT_AFTER(ump->softdep_worklist_tail, wk, wk_list);
 		ump->softdep_worklist_tail = wk;
 	}
 	ump->softdep_on_worklist += 1;
 	if (flags & WK_NODELAY)
 		worklist_speedup(wk->wk_mp);
 }
 
 /*
  * Remove the item to be processed. If we are removing the last
  * item on the list, we need to recalculate the tail pointer.
  */
 static void
 remove_from_worklist(wk)
 	struct worklist *wk;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(wk->wk_mp);
 	WORKLIST_REMOVE(wk);
 	if (ump->softdep_worklist_tail == wk)
 		ump->softdep_worklist_tail =
 		    (struct worklist *)wk->wk_list.le_prev;
 	ump->softdep_on_worklist -= 1;
 }
 
 static void
 wake_worklist(wk)
 	struct worklist *wk;
 {
 	if (wk->wk_state & IOWAITING) {
 		wk->wk_state &= ~IOWAITING;
 		wakeup(wk);
 	}
 }
 
 static void
 wait_worklist(wk, wmesg)
 	struct worklist *wk;
 	char *wmesg;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(wk->wk_mp);
 	wk->wk_state |= IOWAITING;
 	msleep(wk, LOCK_PTR(ump), PVM, wmesg, 0);
 }
 
 /*
  * Process that runs once per second to handle items in the background queue.
  *
  * Note that we ensure that everything is done in the order in which they
  * appear in the queue. The code below depends on this property to ensure
  * that blocks of a file are freed before the inode itself is freed. This
  * ordering ensures that no new <vfsid, inum, lbn> triples will be generated
  * until all the old ones have been purged from the dependency lists.
  */
 static int 
 softdep_process_worklist(mp, full)
 	struct mount *mp;
 	int full;
 {
 	int cnt, matchcnt;
 	struct ufsmount *ump;
 	long starttime;
 
 	KASSERT(mp != NULL, ("softdep_process_worklist: NULL mp"));
 	if (MOUNTEDSOFTDEP(mp) == 0)
 		return (0);
 	matchcnt = 0;
 	ump = VFSTOUFS(mp);
 	ACQUIRE_LOCK(ump);
 	starttime = time_second;
 	softdep_process_journal(mp, NULL, full ? MNT_WAIT : 0);
 	check_clear_deps(mp);
 	while (ump->softdep_on_worklist > 0) {
 		if ((cnt = process_worklist_item(mp, 10, LK_NOWAIT)) == 0)
 			break;
 		else
 			matchcnt += cnt;
 		check_clear_deps(mp);
 		/*
 		 * We do not generally want to stop for buffer space, but if
 		 * we are really being a buffer hog, we will stop and wait.
 		 */
 		if (should_yield()) {
 			FREE_LOCK(ump);
 			kern_yield(PRI_USER);
 			bwillwrite();
 			ACQUIRE_LOCK(ump);
 		}
 		/*
 		 * Never allow processing to run for more than one
 		 * second. This gives the syncer thread the opportunity
 		 * to pause if appropriate.
 		 */
 		if (!full && starttime != time_second)
 			break;
 	}
 	if (full == 0)
 		journal_unsuspend(ump);
 	FREE_LOCK(ump);
 	return (matchcnt);
 }
 
 /*
  * Process all removes associated with a vnode if we are running out of
  * journal space.  Any other process which attempts to flush these will
  * be unable as we have the vnodes locked.
  */
 static void
 process_removes(vp)
 	struct vnode *vp;
 {
 	struct inodedep *inodedep;
 	struct dirrem *dirrem;
 	struct ufsmount *ump;
 	struct mount *mp;
 	ino_t inum;
 
 	mp = vp->v_mount;
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	inum = VTOI(vp)->i_number;
 	for (;;) {
 top:
 		if (inodedep_lookup(mp, inum, 0, &inodedep) == 0)
 			return;
 		LIST_FOREACH(dirrem, &inodedep->id_dirremhd, dm_inonext) {
 			/*
 			 * If another thread is trying to lock this vnode
 			 * it will fail but we must wait for it to do so
 			 * before we can proceed.
 			 */
 			if (dirrem->dm_state & INPROGRESS) {
 				wait_worklist(&dirrem->dm_list, "pwrwait");
 				goto top;
 			}
 			if ((dirrem->dm_state & (COMPLETE | ONWORKLIST)) == 
 			    (COMPLETE | ONWORKLIST))
 				break;
 		}
 		if (dirrem == NULL)
 			return;
 		remove_from_worklist(&dirrem->dm_list);
 		FREE_LOCK(ump);
 		if (vn_start_secondary_write(NULL, &mp, V_NOWAIT))
 			panic("process_removes: suspended filesystem");
 		handle_workitem_remove(dirrem, 0);
 		vn_finished_secondary_write(mp);
 		ACQUIRE_LOCK(ump);
 	}
 }
 
 /*
  * Process all truncations associated with a vnode if we are running out
  * of journal space.  This is called when the vnode lock is already held
  * and no other process can clear the truncation.  This function returns
  * a value greater than zero if it did any work.
  */
 static void
 process_truncates(vp)
 	struct vnode *vp;
 {
 	struct inodedep *inodedep;
 	struct freeblks *freeblks;
 	struct ufsmount *ump;
 	struct mount *mp;
 	ino_t inum;
 	int cgwait;
 
 	mp = vp->v_mount;
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	inum = VTOI(vp)->i_number;
 	for (;;) {
 		if (inodedep_lookup(mp, inum, 0, &inodedep) == 0)
 			return;
 		cgwait = 0;
 		TAILQ_FOREACH(freeblks, &inodedep->id_freeblklst, fb_next) {
 			/* Journal entries not yet written.  */
 			if (!LIST_EMPTY(&freeblks->fb_jblkdephd)) {
 				jwait(&LIST_FIRST(
 				    &freeblks->fb_jblkdephd)->jb_list,
 				    MNT_WAIT);
 				break;
 			}
 			/* Another thread is executing this item. */
 			if (freeblks->fb_state & INPROGRESS) {
 				wait_worklist(&freeblks->fb_list, "ptrwait");
 				break;
 			}
 			/* Freeblks is waiting on a inode write. */
 			if ((freeblks->fb_state & COMPLETE) == 0) {
 				FREE_LOCK(ump);
 				ffs_update(vp, 1);
 				ACQUIRE_LOCK(ump);
 				break;
 			}
 			if ((freeblks->fb_state & (ALLCOMPLETE | ONWORKLIST)) ==
 			    (ALLCOMPLETE | ONWORKLIST)) {
 				remove_from_worklist(&freeblks->fb_list);
 				freeblks->fb_state |= INPROGRESS;
 				FREE_LOCK(ump);
 				if (vn_start_secondary_write(NULL, &mp,
 				    V_NOWAIT))
 					panic("process_truncates: "
 					    "suspended filesystem");
 				handle_workitem_freeblocks(freeblks, 0);
 				vn_finished_secondary_write(mp);
 				ACQUIRE_LOCK(ump);
 				break;
 			}
 			if (freeblks->fb_cgwait)
 				cgwait++;
 		}
 		if (cgwait) {
 			FREE_LOCK(ump);
 			sync_cgs(mp, MNT_WAIT);
 			ffs_sync_snap(mp, MNT_WAIT);
 			ACQUIRE_LOCK(ump);
 			continue;
 		}
 		if (freeblks == NULL)
 			break;
 	}
 	return;
 }
 
 /*
  * Process one item on the worklist.
  */
 static int
 process_worklist_item(mp, target, flags)
 	struct mount *mp;
 	int target;
 	int flags;
 {
 	struct worklist sentinel;
 	struct worklist *wk;
 	struct ufsmount *ump;
 	int matchcnt;
 	int error;
 
 	KASSERT(mp != NULL, ("process_worklist_item: NULL mp"));
 	/*
 	 * If we are being called because of a process doing a
 	 * copy-on-write, then it is not safe to write as we may
 	 * recurse into the copy-on-write routine.
 	 */
 	if (curthread->td_pflags & TDP_COWINPROGRESS)
 		return (-1);
 	PHOLD(curproc);	/* Don't let the stack go away. */
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	matchcnt = 0;
 	sentinel.wk_mp = NULL;
 	sentinel.wk_type = D_SENTINEL;
 	LIST_INSERT_HEAD(&ump->softdep_workitem_pending, &sentinel, wk_list);
 	for (wk = LIST_NEXT(&sentinel, wk_list); wk != NULL;
 	    wk = LIST_NEXT(&sentinel, wk_list)) {
 		if (wk->wk_type == D_SENTINEL) {
 			LIST_REMOVE(&sentinel, wk_list);
 			LIST_INSERT_AFTER(wk, &sentinel, wk_list);
 			continue;
 		}
 		if (wk->wk_state & INPROGRESS)
 			panic("process_worklist_item: %p already in progress.",
 			    wk);
 		wk->wk_state |= INPROGRESS;
 		remove_from_worklist(wk);
 		FREE_LOCK(ump);
 		if (vn_start_secondary_write(NULL, &mp, V_NOWAIT))
 			panic("process_worklist_item: suspended filesystem");
 		switch (wk->wk_type) {
 		case D_DIRREM:
 			/* removal of a directory entry */
 			error = handle_workitem_remove(WK_DIRREM(wk), flags);
 			break;
 
 		case D_FREEBLKS:
 			/* releasing blocks and/or fragments from a file */
 			error = handle_workitem_freeblocks(WK_FREEBLKS(wk),
 			    flags);
 			break;
 
 		case D_FREEFRAG:
 			/* releasing a fragment when replaced as a file grows */
 			handle_workitem_freefrag(WK_FREEFRAG(wk));
 			error = 0;
 			break;
 
 		case D_FREEFILE:
 			/* releasing an inode when its link count drops to 0 */
 			handle_workitem_freefile(WK_FREEFILE(wk));
 			error = 0;
 			break;
 
 		default:
 			panic("%s_process_worklist: Unknown type %s",
 			    "softdep", TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 		vn_finished_secondary_write(mp);
 		ACQUIRE_LOCK(ump);
 		if (error == 0) {
 			if (++matchcnt == target)
 				break;
 			continue;
 		}
 		/*
 		 * We have to retry the worklist item later.  Wake up any
 		 * waiters who may be able to complete it immediately and
 		 * add the item back to the head so we don't try to execute
 		 * it again.
 		 */
 		wk->wk_state &= ~INPROGRESS;
 		wake_worklist(wk);
 		add_to_worklist(wk, WK_HEAD);
 	}
 	LIST_REMOVE(&sentinel, wk_list);
 	/* Sentinal could've become the tail from remove_from_worklist. */
 	if (ump->softdep_worklist_tail == &sentinel)
 		ump->softdep_worklist_tail =
 		    (struct worklist *)sentinel.wk_list.le_prev;
 	PRELE(curproc);
 	return (matchcnt);
 }
 
 /*
  * Move dependencies from one buffer to another.
  */
 int
 softdep_move_dependencies(oldbp, newbp)
 	struct buf *oldbp;
 	struct buf *newbp;
 {
 	struct worklist *wk, *wktail;
 	struct ufsmount *ump;
 	int dirty;
 
 	if ((wk = LIST_FIRST(&oldbp->b_dep)) == NULL)
 		return (0);
 	KASSERT(MOUNTEDSOFTDEP(wk->wk_mp) != 0,
 	    ("softdep_move_dependencies called on non-softdep filesystem"));
 	dirty = 0;
 	wktail = NULL;
 	ump = VFSTOUFS(wk->wk_mp);
 	ACQUIRE_LOCK(ump);
 	while ((wk = LIST_FIRST(&oldbp->b_dep)) != NULL) {
 		LIST_REMOVE(wk, wk_list);
 		if (wk->wk_type == D_BMSAFEMAP &&
 		    bmsafemap_backgroundwrite(WK_BMSAFEMAP(wk), newbp))
 			dirty = 1;
 		if (wktail == 0)
 			LIST_INSERT_HEAD(&newbp->b_dep, wk, wk_list);
 		else
 			LIST_INSERT_AFTER(wktail, wk, wk_list);
 		wktail = wk;
 	}
 	FREE_LOCK(ump);
 
 	return (dirty);
 }
 
 /*
  * Purge the work list of all items associated with a particular mount point.
  */
 int
 softdep_flushworklist(oldmnt, countp, td)
 	struct mount *oldmnt;
 	int *countp;
 	struct thread *td;
 {
 	struct vnode *devvp;
 	int count, error = 0;
 	struct ufsmount *ump;
 
 	/*
 	 * Alternately flush the block device associated with the mount
 	 * point and process any dependencies that the flushing
 	 * creates. We continue until no more worklist dependencies
 	 * are found.
 	 */
 	*countp = 0;
 	ump = VFSTOUFS(oldmnt);
 	devvp = ump->um_devvp;
 	while ((count = softdep_process_worklist(oldmnt, 1)) > 0) {
 		*countp += count;
 		vn_lock(devvp, LK_EXCLUSIVE | LK_RETRY);
 		error = VOP_FSYNC(devvp, MNT_WAIT, td);
 		VOP_UNLOCK(devvp, 0);
 		if (error)
 			break;
 	}
 	return (error);
 }
 
 static int
 softdep_waitidle(struct mount *mp, int flags __unused)
 {
 	struct ufsmount *ump;
 	int error;
 	int i;
 
 	ump = VFSTOUFS(mp);
 	ACQUIRE_LOCK(ump);
 	for (i = 0; i < 10 && ump->softdep_deps; i++) {
 		ump->softdep_req = 1;
 		KASSERT((flags & FORCECLOSE) == 0 ||
 		    ump->softdep_on_worklist == 0,
 		    ("softdep_waitidle: work added after flush"));
 		msleep(&ump->softdep_deps, LOCK_PTR(ump), PVM, "softdeps", 1);
 	}
 	ump->softdep_req = 0;
 	FREE_LOCK(ump);
 	error = 0;
 	if (i == 10) {
 		error = EBUSY;
 		printf("softdep_waitidle: Failed to flush worklist for %p\n",
 		    mp);
 	}
 
 	return (error);
 }
 
 /*
  * Flush all vnodes and worklist items associated with a specified mount point.
  */
 int
 softdep_flushfiles(oldmnt, flags, td)
 	struct mount *oldmnt;
 	int flags;
 	struct thread *td;
 {
 #ifdef QUOTA
 	struct ufsmount *ump;
 	int i;
 #endif
 	int error, early, depcount, loopcnt, retry_flush_count, retry;
 	int morework;
 
 	KASSERT(MOUNTEDSOFTDEP(oldmnt) != 0,
 	    ("softdep_flushfiles called on non-softdep filesystem"));
 	loopcnt = 10;
 	retry_flush_count = 3;
 retry_flush:
 	error = 0;
 
 	/*
 	 * Alternately flush the vnodes associated with the mount
 	 * point and process any dependencies that the flushing
 	 * creates. In theory, this loop can happen at most twice,
 	 * but we give it a few extra just to be sure.
 	 */
 	for (; loopcnt > 0; loopcnt--) {
 		/*
 		 * Do another flush in case any vnodes were brought in
 		 * as part of the cleanup operations.
 		 */
 		early = retry_flush_count == 1 || (oldmnt->mnt_kern_flag &
 		    MNTK_UNMOUNT) == 0 ? 0 : EARLYFLUSH;
 		if ((error = ffs_flushfiles(oldmnt, flags | early, td)) != 0)
 			break;
 		if ((error = softdep_flushworklist(oldmnt, &depcount, td)) != 0 ||
 		    depcount == 0)
 			break;
 	}
 	/*
 	 * If we are unmounting then it is an error to fail. If we
 	 * are simply trying to downgrade to read-only, then filesystem
 	 * activity can keep us busy forever, so we just fail with EBUSY.
 	 */
 	if (loopcnt == 0) {
 		if (oldmnt->mnt_kern_flag & MNTK_UNMOUNT)
 			panic("softdep_flushfiles: looping");
 		error = EBUSY;
 	}
 	if (!error)
 		error = softdep_waitidle(oldmnt, flags);
 	if (!error) {
 		if (oldmnt->mnt_kern_flag & MNTK_UNMOUNT) {
 			retry = 0;
 			MNT_ILOCK(oldmnt);
 			KASSERT((oldmnt->mnt_kern_flag & MNTK_NOINSMNTQ) != 0,
 			    ("softdep_flushfiles: !MNTK_NOINSMNTQ"));
 			morework = oldmnt->mnt_nvnodelistsize > 0;
 #ifdef QUOTA
 			ump = VFSTOUFS(oldmnt);
 			UFS_LOCK(ump);
 			for (i = 0; i < MAXQUOTAS; i++) {
 				if (ump->um_quotas[i] != NULLVP)
 					morework = 1;
 			}
 			UFS_UNLOCK(ump);
 #endif
 			if (morework) {
 				if (--retry_flush_count > 0) {
 					retry = 1;
 					loopcnt = 3;
 				} else
 					error = EBUSY;
 			}
 			MNT_IUNLOCK(oldmnt);
 			if (retry)
 				goto retry_flush;
 		}
 	}
 	return (error);
 }
 
 /*
  * Structure hashing.
  * 
  * There are four types of structures that can be looked up:
  *	1) pagedep structures identified by mount point, inode number,
  *	   and logical block.
  *	2) inodedep structures identified by mount point and inode number.
  *	3) newblk structures identified by mount point and
  *	   physical block number.
  *	4) bmsafemap structures identified by mount point and
  *	   cylinder group number.
  *
  * The "pagedep" and "inodedep" dependency structures are hashed
  * separately from the file blocks and inodes to which they correspond.
  * This separation helps when the in-memory copy of an inode or
  * file block must be replaced. It also obviates the need to access
  * an inode or file page when simply updating (or de-allocating)
  * dependency structures. Lookup of newblk structures is needed to
  * find newly allocated blocks when trying to associate them with
  * their allocdirect or allocindir structure.
  *
  * The lookup routines optionally create and hash a new instance when
  * an existing entry is not found. The bmsafemap lookup routine always
  * allocates a new structure if an existing one is not found.
  */
 #define DEPALLOC	0x0001	/* allocate structure if lookup fails */
 #define NODELAY		0x0002	/* cannot do background work */
 
 /*
  * Structures and routines associated with pagedep caching.
  */
 #define	PAGEDEP_HASH(ump, inum, lbn) \
 	(&(ump)->pagedep_hashtbl[((inum) + (lbn)) & (ump)->pagedep_hash_size])
 
 static int
 pagedep_find(pagedephd, ino, lbn, pagedeppp)
 	struct pagedep_hashhead *pagedephd;
 	ino_t ino;
 	ufs_lbn_t lbn;
 	struct pagedep **pagedeppp;
 {
 	struct pagedep *pagedep;
 
 	LIST_FOREACH(pagedep, pagedephd, pd_hash) {
 		if (ino == pagedep->pd_ino && lbn == pagedep->pd_lbn) {
 			*pagedeppp = pagedep;
 			return (1);
 		}
 	}
 	*pagedeppp = NULL;
 	return (0);
 }
 /*
  * Look up a pagedep. Return 1 if found, 0 otherwise.
  * If not found, allocate if DEPALLOC flag is passed.
  * Found or allocated entry is returned in pagedeppp.
  * This routine must be called with splbio interrupts blocked.
  */
 static int
 pagedep_lookup(mp, bp, ino, lbn, flags, pagedeppp)
 	struct mount *mp;
 	struct buf *bp;
 	ino_t ino;
 	ufs_lbn_t lbn;
 	int flags;
 	struct pagedep **pagedeppp;
 {
 	struct pagedep *pagedep;
 	struct pagedep_hashhead *pagedephd;
 	struct worklist *wk;
 	struct ufsmount *ump;
 	int ret;
 	int i;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	if (bp) {
 		LIST_FOREACH(wk, &bp->b_dep, wk_list) {
 			if (wk->wk_type == D_PAGEDEP) {
 				*pagedeppp = WK_PAGEDEP(wk);
 				return (1);
 			}
 		}
 	}
 	pagedephd = PAGEDEP_HASH(ump, ino, lbn);
 	ret = pagedep_find(pagedephd, ino, lbn, pagedeppp);
 	if (ret) {
 		if (((*pagedeppp)->pd_state & ONWORKLIST) == 0 && bp)
 			WORKLIST_INSERT(&bp->b_dep, &(*pagedeppp)->pd_list);
 		return (1);
 	}
 	if ((flags & DEPALLOC) == 0)
 		return (0);
 	FREE_LOCK(ump);
 	pagedep = malloc(sizeof(struct pagedep),
 	    M_PAGEDEP, M_SOFTDEP_FLAGS|M_ZERO);
 	workitem_alloc(&pagedep->pd_list, D_PAGEDEP, mp);
 	ACQUIRE_LOCK(ump);
 	ret = pagedep_find(pagedephd, ino, lbn, pagedeppp);
 	if (*pagedeppp) {
 		/*
 		 * This should never happen since we only create pagedeps
 		 * with the vnode lock held.  Could be an assert.
 		 */
 		WORKITEM_FREE(pagedep, D_PAGEDEP);
 		return (ret);
 	}
 	pagedep->pd_ino = ino;
 	pagedep->pd_lbn = lbn;
 	LIST_INIT(&pagedep->pd_dirremhd);
 	LIST_INIT(&pagedep->pd_pendinghd);
 	for (i = 0; i < DAHASHSZ; i++)
 		LIST_INIT(&pagedep->pd_diraddhd[i]);
 	LIST_INSERT_HEAD(pagedephd, pagedep, pd_hash);
 	WORKLIST_INSERT(&bp->b_dep, &pagedep->pd_list);
 	*pagedeppp = pagedep;
 	return (0);
 }
 
 /*
  * Structures and routines associated with inodedep caching.
  */
 #define	INODEDEP_HASH(ump, inum) \
       (&(ump)->inodedep_hashtbl[(inum) & (ump)->inodedep_hash_size])
 
 static int
 inodedep_find(inodedephd, inum, inodedeppp)
 	struct inodedep_hashhead *inodedephd;
 	ino_t inum;
 	struct inodedep **inodedeppp;
 {
 	struct inodedep *inodedep;
 
 	LIST_FOREACH(inodedep, inodedephd, id_hash)
 		if (inum == inodedep->id_ino)
 			break;
 	if (inodedep) {
 		*inodedeppp = inodedep;
 		return (1);
 	}
 	*inodedeppp = NULL;
 
 	return (0);
 }
 /*
  * Look up an inodedep. Return 1 if found, 0 if not found.
  * If not found, allocate if DEPALLOC flag is passed.
  * Found or allocated entry is returned in inodedeppp.
  * This routine must be called with splbio interrupts blocked.
  */
 static int
 inodedep_lookup(mp, inum, flags, inodedeppp)
 	struct mount *mp;
 	ino_t inum;
 	int flags;
 	struct inodedep **inodedeppp;
 {
 	struct inodedep *inodedep;
 	struct inodedep_hashhead *inodedephd;
 	struct ufsmount *ump;
 	struct fs *fs;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	fs = ump->um_fs;
 	inodedephd = INODEDEP_HASH(ump, inum);
 
 	if (inodedep_find(inodedephd, inum, inodedeppp))
 		return (1);
 	if ((flags & DEPALLOC) == 0)
 		return (0);
 	/*
 	 * If the system is over its limit and our filesystem is
 	 * responsible for more than our share of that usage and
 	 * we are not in a rush, request some inodedep cleanup.
 	 */
 	while (dep_current[D_INODEDEP] > max_softdeps &&
 	    (flags & NODELAY) == 0 &&
 	    ump->softdep_curdeps[D_INODEDEP] >
 	    max_softdeps / stat_flush_threads)
 		request_cleanup(mp, FLUSH_INODES);
 	FREE_LOCK(ump);
 	inodedep = malloc(sizeof(struct inodedep),
 		M_INODEDEP, M_SOFTDEP_FLAGS);
 	workitem_alloc(&inodedep->id_list, D_INODEDEP, mp);
 	ACQUIRE_LOCK(ump);
 	if (inodedep_find(inodedephd, inum, inodedeppp)) {
 		WORKITEM_FREE(inodedep, D_INODEDEP);
 		return (1);
 	}
 	inodedep->id_fs = fs;
 	inodedep->id_ino = inum;
 	inodedep->id_state = ALLCOMPLETE;
 	inodedep->id_nlinkdelta = 0;
 	inodedep->id_savedino1 = NULL;
 	inodedep->id_savedsize = -1;
 	inodedep->id_savedextsize = -1;
 	inodedep->id_savednlink = -1;
 	inodedep->id_bmsafemap = NULL;
 	inodedep->id_mkdiradd = NULL;
 	LIST_INIT(&inodedep->id_dirremhd);
 	LIST_INIT(&inodedep->id_pendinghd);
 	LIST_INIT(&inodedep->id_inowait);
 	LIST_INIT(&inodedep->id_bufwait);
 	TAILQ_INIT(&inodedep->id_inoreflst);
 	TAILQ_INIT(&inodedep->id_inoupdt);
 	TAILQ_INIT(&inodedep->id_newinoupdt);
 	TAILQ_INIT(&inodedep->id_extupdt);
 	TAILQ_INIT(&inodedep->id_newextupdt);
 	TAILQ_INIT(&inodedep->id_freeblklst);
 	LIST_INSERT_HEAD(inodedephd, inodedep, id_hash);
 	*inodedeppp = inodedep;
 	return (0);
 }
 
 /*
  * Structures and routines associated with newblk caching.
  */
 #define	NEWBLK_HASH(ump, inum) \
 	(&(ump)->newblk_hashtbl[(inum) & (ump)->newblk_hash_size])
 
 static int
 newblk_find(newblkhd, newblkno, flags, newblkpp)
 	struct newblk_hashhead *newblkhd;
 	ufs2_daddr_t newblkno;
 	int flags;
 	struct newblk **newblkpp;
 {
 	struct newblk *newblk;
 
 	LIST_FOREACH(newblk, newblkhd, nb_hash) {
 		if (newblkno != newblk->nb_newblkno)
 			continue;
 		/*
 		 * If we're creating a new dependency don't match those that
 		 * have already been converted to allocdirects.  This is for
 		 * a frag extend.
 		 */
 		if ((flags & DEPALLOC) && newblk->nb_list.wk_type != D_NEWBLK)
 			continue;
 		break;
 	}
 	if (newblk) {
 		*newblkpp = newblk;
 		return (1);
 	}
 	*newblkpp = NULL;
 	return (0);
 }
 
 /*
  * Look up a newblk. Return 1 if found, 0 if not found.
  * If not found, allocate if DEPALLOC flag is passed.
  * Found or allocated entry is returned in newblkpp.
  */
 static int
 newblk_lookup(mp, newblkno, flags, newblkpp)
 	struct mount *mp;
 	ufs2_daddr_t newblkno;
 	int flags;
 	struct newblk **newblkpp;
 {
 	struct newblk *newblk;
 	struct newblk_hashhead *newblkhd;
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	newblkhd = NEWBLK_HASH(ump, newblkno);
 	if (newblk_find(newblkhd, newblkno, flags, newblkpp))
 		return (1);
 	if ((flags & DEPALLOC) == 0)
 		return (0);
 	FREE_LOCK(ump);
 	newblk = malloc(sizeof(union allblk), M_NEWBLK,
 	    M_SOFTDEP_FLAGS | M_ZERO);
 	workitem_alloc(&newblk->nb_list, D_NEWBLK, mp);
 	ACQUIRE_LOCK(ump);
 	if (newblk_find(newblkhd, newblkno, flags, newblkpp)) {
 		WORKITEM_FREE(newblk, D_NEWBLK);
 		return (1);
 	}
 	newblk->nb_freefrag = NULL;
 	LIST_INIT(&newblk->nb_indirdeps);
 	LIST_INIT(&newblk->nb_newdirblk);
 	LIST_INIT(&newblk->nb_jwork);
 	newblk->nb_state = ATTACHED;
 	newblk->nb_newblkno = newblkno;
 	LIST_INSERT_HEAD(newblkhd, newblk, nb_hash);
 	*newblkpp = newblk;
 	return (0);
 }
 
 /*
  * Structures and routines associated with freed indirect block caching.
  */
 #define	INDIR_HASH(ump, blkno) \
 	(&(ump)->indir_hashtbl[(blkno) & (ump)->indir_hash_size])
 
 /*
  * Lookup an indirect block in the indir hash table.  The freework is
  * removed and potentially freed.  The caller must do a blocking journal
  * write before writing to the blkno.
  */
 static int
 indirblk_lookup(mp, blkno)
 	struct mount *mp;
 	ufs2_daddr_t blkno;
 {
 	struct freework *freework;
 	struct indir_hashhead *wkhd;
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(mp);
 	wkhd = INDIR_HASH(ump, blkno);
 	TAILQ_FOREACH(freework, wkhd, fw_next) {
 		if (freework->fw_blkno != blkno)
 			continue;
 		indirblk_remove(freework);
 		return (1);
 	}
 	return (0);
 }
 
 /*
  * Insert an indirect block represented by freework into the indirblk
  * hash table so that it may prevent the block from being re-used prior
  * to the journal being written.
  */
 static void
 indirblk_insert(freework)
 	struct freework *freework;
 {
 	struct jblocks *jblocks;
 	struct jseg *jseg;
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(freework->fw_list.wk_mp);
 	jblocks = ump->softdep_jblocks;
 	jseg = TAILQ_LAST(&jblocks->jb_segs, jseglst);
 	if (jseg == NULL)
 		return;
 	
 	LIST_INSERT_HEAD(&jseg->js_indirs, freework, fw_segs);
 	TAILQ_INSERT_HEAD(INDIR_HASH(ump, freework->fw_blkno), freework,
 	    fw_next);
 	freework->fw_state &= ~DEPCOMPLETE;
 }
 
 static void
 indirblk_remove(freework)
 	struct freework *freework;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(freework->fw_list.wk_mp);
 	LIST_REMOVE(freework, fw_segs);
 	TAILQ_REMOVE(INDIR_HASH(ump, freework->fw_blkno), freework, fw_next);
 	freework->fw_state |= DEPCOMPLETE;
 	if ((freework->fw_state & ALLCOMPLETE) == ALLCOMPLETE)
 		WORKITEM_FREE(freework, D_FREEWORK);
 }
 
 /*
  * Executed during filesystem system initialization before
  * mounting any filesystems.
  */
 void 
 softdep_initialize()
 {
 
 	TAILQ_INIT(&softdepmounts);
 	max_softdeps = desiredvnodes * 4;
 
 	/* initialise bioops hack */
 	bioops.io_start = softdep_disk_io_initiation;
 	bioops.io_complete = softdep_disk_write_complete;
 	bioops.io_deallocate = softdep_deallocate_dependencies;
 	bioops.io_countdeps = softdep_count_dependencies;
 
 	/* Initialize the callout with an mtx. */
 	callout_init_mtx(&softdep_callout, &lk, 0);
 }
 
 /*
  * Executed after all filesystems have been unmounted during
  * filesystem module unload.
  */
 void
 softdep_uninitialize()
 {
 
 	/* clear bioops hack */
 	bioops.io_start = NULL;
 	bioops.io_complete = NULL;
 	bioops.io_deallocate = NULL;
 	bioops.io_countdeps = NULL;
 
 	callout_drain(&softdep_callout);
 }
 
 /*
  * Called at mount time to notify the dependency code that a
  * filesystem wishes to use it.
  */
 int
 softdep_mount(devvp, mp, fs, cred)
 	struct vnode *devvp;
 	struct mount *mp;
 	struct fs *fs;
 	struct ucred *cred;
 {
 	struct csum_total cstotal;
 	struct mount_softdeps *sdp;
 	struct ufsmount *ump;
 	struct cg *cgp;
 	struct buf *bp;
 	int i, error, cyl;
 
 	sdp = malloc(sizeof(struct mount_softdeps), M_MOUNTDATA,
 	    M_WAITOK | M_ZERO);
 	MNT_ILOCK(mp);
 	mp->mnt_flag = (mp->mnt_flag & ~MNT_ASYNC) | MNT_SOFTDEP;
 	if ((mp->mnt_kern_flag & MNTK_SOFTDEP) == 0) {
 		mp->mnt_kern_flag = (mp->mnt_kern_flag & ~MNTK_ASYNC) | 
 			MNTK_SOFTDEP | MNTK_NOASYNC;
 	}
 	ump = VFSTOUFS(mp);
 	ump->um_softdep = sdp;
 	MNT_IUNLOCK(mp);
 	rw_init(LOCK_PTR(ump), "Per-Filesystem Softdep Lock");
 	sdp->sd_ump = ump;
 	LIST_INIT(&ump->softdep_workitem_pending);
 	LIST_INIT(&ump->softdep_journal_pending);
 	TAILQ_INIT(&ump->softdep_unlinked);
 	LIST_INIT(&ump->softdep_dirtycg);
 	ump->softdep_worklist_tail = NULL;
 	ump->softdep_on_worklist = 0;
 	ump->softdep_deps = 0;
 	LIST_INIT(&ump->softdep_mkdirlisthd);
 	ump->pagedep_hashtbl = hashinit(desiredvnodes / 5, M_PAGEDEP,
 	    &ump->pagedep_hash_size);
 	ump->pagedep_nextclean = 0;
 	ump->inodedep_hashtbl = hashinit(desiredvnodes, M_INODEDEP,
 	    &ump->inodedep_hash_size);
 	ump->inodedep_nextclean = 0;
 	ump->newblk_hashtbl = hashinit(max_softdeps / 2,  M_NEWBLK,
 	    &ump->newblk_hash_size);
 	ump->bmsafemap_hashtbl = hashinit(1024, M_BMSAFEMAP,
 	    &ump->bmsafemap_hash_size);
 	i = 1 << (ffs(desiredvnodes / 10) - 1);
 	ump->indir_hashtbl = malloc(i * sizeof(struct indir_hashhead),
 	    M_FREEWORK, M_WAITOK);
 	ump->indir_hash_size = i - 1;
 	for (i = 0; i <= ump->indir_hash_size; i++)
 		TAILQ_INIT(&ump->indir_hashtbl[i]);
 	ACQUIRE_GBLLOCK(&lk);
 	TAILQ_INSERT_TAIL(&softdepmounts, sdp, sd_next);
 	FREE_GBLLOCK(&lk);
 	if ((fs->fs_flags & FS_SUJ) &&
 	    (error = journal_mount(mp, fs, cred)) != 0) {
 		printf("Failed to start journal: %d\n", error);
 		softdep_unmount(mp);
 		return (error);
 	}
 	/*
 	 * Start our flushing thread in the bufdaemon process.
 	 */
+	ACQUIRE_LOCK(ump);
+	ump->softdep_flags |= FLUSH_STARTING;
+	FREE_LOCK(ump);
 	kproc_kthread_add(&softdep_flush, mp, &bufdaemonproc,
 	    &ump->softdep_flushtd, 0, 0, "softdepflush", "%s worker",
 	    mp->mnt_stat.f_mntonname);
+	ACQUIRE_LOCK(ump);
+	while ((ump->softdep_flags & FLUSH_STARTING) != 0) {
+		msleep(&ump->softdep_flushtd, LOCK_PTR(ump), PVM, "sdstart",
+		    hz / 2);
+	}
+	FREE_LOCK(ump);
 	/*
 	 * When doing soft updates, the counters in the
 	 * superblock may have gotten out of sync. Recomputation
 	 * can take a long time and can be deferred for background
 	 * fsck.  However, the old behavior of scanning the cylinder
 	 * groups and recalculating them at mount time is available
 	 * by setting vfs.ffs.compute_summary_at_mount to one.
 	 */
 	if (compute_summary_at_mount == 0 || fs->fs_clean != 0)
 		return (0);
 	bzero(&cstotal, sizeof cstotal);
 	for (cyl = 0; cyl < fs->fs_ncg; cyl++) {
 		if ((error = bread(devvp, fsbtodb(fs, cgtod(fs, cyl)),
 		    fs->fs_cgsize, cred, &bp)) != 0) {
 			brelse(bp);
 			softdep_unmount(mp);
 			return (error);
 		}
 		cgp = (struct cg *)bp->b_data;
 		cstotal.cs_nffree += cgp->cg_cs.cs_nffree;
 		cstotal.cs_nbfree += cgp->cg_cs.cs_nbfree;
 		cstotal.cs_nifree += cgp->cg_cs.cs_nifree;
 		cstotal.cs_ndir += cgp->cg_cs.cs_ndir;
 		fs->fs_cs(fs, cyl) = cgp->cg_cs;
 		brelse(bp);
 	}
 #ifdef DEBUG
 	if (bcmp(&cstotal, &fs->fs_cstotal, sizeof cstotal))
 		printf("%s: superblock summary recomputed\n", fs->fs_fsmnt);
 #endif
 	bcopy(&cstotal, &fs->fs_cstotal, sizeof cstotal);
 	return (0);
 }
 
 void
 softdep_unmount(mp)
 	struct mount *mp;
 {
 	struct ufsmount *ump;
 #ifdef INVARIANTS
 	int i;
 #endif
 
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_unmount called on non-softdep filesystem"));
 	ump = VFSTOUFS(mp);
 	MNT_ILOCK(mp);
 	mp->mnt_flag &= ~MNT_SOFTDEP;
 	if (MOUNTEDSUJ(mp) == 0) {
 		MNT_IUNLOCK(mp);
 	} else {
 		mp->mnt_flag &= ~MNT_SUJ;
 		MNT_IUNLOCK(mp);
 		journal_unmount(ump);
 	}
 	/*
 	 * Shut down our flushing thread. Check for NULL is if
 	 * softdep_mount errors out before the thread has been created.
 	 */
 	if (ump->softdep_flushtd != NULL) {
 		ACQUIRE_LOCK(ump);
 		ump->softdep_flags |= FLUSH_EXIT;
 		wakeup(&ump->softdep_flushtd);
 		msleep(&ump->softdep_flags, LOCK_PTR(ump), PVM | PDROP,
 		    "sdwait", 0);
 		KASSERT((ump->softdep_flags & FLUSH_EXIT) == 0,
 		    ("Thread shutdown failed"));
 	}
 	/*
 	 * Free up our resources.
 	 */
 	ACQUIRE_GBLLOCK(&lk);
 	TAILQ_REMOVE(&softdepmounts, ump->um_softdep, sd_next);
 	FREE_GBLLOCK(&lk);
 	rw_destroy(LOCK_PTR(ump));
 	hashdestroy(ump->pagedep_hashtbl, M_PAGEDEP, ump->pagedep_hash_size);
 	hashdestroy(ump->inodedep_hashtbl, M_INODEDEP, ump->inodedep_hash_size);
 	hashdestroy(ump->newblk_hashtbl, M_NEWBLK, ump->newblk_hash_size);
 	hashdestroy(ump->bmsafemap_hashtbl, M_BMSAFEMAP,
 	    ump->bmsafemap_hash_size);
 	free(ump->indir_hashtbl, M_FREEWORK);
 #ifdef INVARIANTS
 	for (i = 0; i <= D_LAST; i++)
 		KASSERT(ump->softdep_curdeps[i] == 0,
 		    ("Unmount %s: Dep type %s != 0 (%ld)", ump->um_fs->fs_fsmnt,
 		    TYPENAME(i), ump->softdep_curdeps[i]));
 #endif
 	free(ump->um_softdep, M_MOUNTDATA);
 }
 
 static struct jblocks *
 jblocks_create(void)
 {
 	struct jblocks *jblocks;
 
 	jblocks = malloc(sizeof(*jblocks), M_JBLOCKS, M_WAITOK | M_ZERO);
 	TAILQ_INIT(&jblocks->jb_segs);
 	jblocks->jb_avail = 10;
 	jblocks->jb_extent = malloc(sizeof(struct jextent) * jblocks->jb_avail,
 	    M_JBLOCKS, M_WAITOK | M_ZERO);
 
 	return (jblocks);
 }
 
 static ufs2_daddr_t
 jblocks_alloc(jblocks, bytes, actual)
 	struct jblocks *jblocks;
 	int bytes;
 	int *actual;
 {
 	ufs2_daddr_t daddr;
 	struct jextent *jext;
 	int freecnt;
 	int blocks;
 
 	blocks = bytes / DEV_BSIZE;
 	jext = &jblocks->jb_extent[jblocks->jb_head];
 	freecnt = jext->je_blocks - jblocks->jb_off;
 	if (freecnt == 0) {
 		jblocks->jb_off = 0;
 		if (++jblocks->jb_head > jblocks->jb_used)
 			jblocks->jb_head = 0;
 		jext = &jblocks->jb_extent[jblocks->jb_head];
 		freecnt = jext->je_blocks;
 	}
 	if (freecnt > blocks)
 		freecnt = blocks;
 	*actual = freecnt * DEV_BSIZE;
 	daddr = jext->je_daddr + jblocks->jb_off;
 	jblocks->jb_off += freecnt;
 	jblocks->jb_free -= freecnt;
 
 	return (daddr);
 }
 
 static void
 jblocks_free(jblocks, mp, bytes)
 	struct jblocks *jblocks;
 	struct mount *mp;
 	int bytes;
 {
 
 	LOCK_OWNED(VFSTOUFS(mp));
 	jblocks->jb_free += bytes / DEV_BSIZE;
 	if (jblocks->jb_suspended)
 		worklist_speedup(mp);
 	wakeup(jblocks);
 }
 
 static void
 jblocks_destroy(jblocks)
 	struct jblocks *jblocks;
 {
 
 	if (jblocks->jb_extent)
 		free(jblocks->jb_extent, M_JBLOCKS);
 	free(jblocks, M_JBLOCKS);
 }
 
 static void
 jblocks_add(jblocks, daddr, blocks)
 	struct jblocks *jblocks;
 	ufs2_daddr_t daddr;
 	int blocks;
 {
 	struct jextent *jext;
 
 	jblocks->jb_blocks += blocks;
 	jblocks->jb_free += blocks;
 	jext = &jblocks->jb_extent[jblocks->jb_used];
 	/* Adding the first block. */
 	if (jext->je_daddr == 0) {
 		jext->je_daddr = daddr;
 		jext->je_blocks = blocks;
 		return;
 	}
 	/* Extending the last extent. */
 	if (jext->je_daddr + jext->je_blocks == daddr) {
 		jext->je_blocks += blocks;
 		return;
 	}
 	/* Adding a new extent. */
 	if (++jblocks->jb_used == jblocks->jb_avail) {
 		jblocks->jb_avail *= 2;
 		jext = malloc(sizeof(struct jextent) * jblocks->jb_avail,
 		    M_JBLOCKS, M_WAITOK | M_ZERO);
 		memcpy(jext, jblocks->jb_extent,
 		    sizeof(struct jextent) * jblocks->jb_used);
 		free(jblocks->jb_extent, M_JBLOCKS);
 		jblocks->jb_extent = jext;
 	}
 	jext = &jblocks->jb_extent[jblocks->jb_used];
 	jext->je_daddr = daddr;
 	jext->je_blocks = blocks;
 	return;
 }
 
 int
 softdep_journal_lookup(mp, vpp)
 	struct mount *mp;
 	struct vnode **vpp;
 {
 	struct componentname cnp;
 	struct vnode *dvp;
 	ino_t sujournal;
 	int error;
 
 	error = VFS_VGET(mp, ROOTINO, LK_EXCLUSIVE, &dvp);
 	if (error)
 		return (error);
 	bzero(&cnp, sizeof(cnp));
 	cnp.cn_nameiop = LOOKUP;
 	cnp.cn_flags = ISLASTCN;
 	cnp.cn_thread = curthread;
 	cnp.cn_cred = curthread->td_ucred;
 	cnp.cn_pnbuf = SUJ_FILE;
 	cnp.cn_nameptr = SUJ_FILE;
 	cnp.cn_namelen = strlen(SUJ_FILE);
 	error = ufs_lookup_ino(dvp, NULL, &cnp, &sujournal);
 	vput(dvp);
 	if (error != 0)
 		return (error);
 	error = VFS_VGET(mp, sujournal, LK_EXCLUSIVE, vpp);
 	return (error);
 }
 
 /*
  * Open and verify the journal file.
  */
 static int
 journal_mount(mp, fs, cred)
 	struct mount *mp;
 	struct fs *fs;
 	struct ucred *cred;
 {
 	struct jblocks *jblocks;
 	struct ufsmount *ump;
 	struct vnode *vp;
 	struct inode *ip;
 	ufs2_daddr_t blkno;
 	int bcount;
 	int error;
 	int i;
 
 	ump = VFSTOUFS(mp);
 	ump->softdep_journal_tail = NULL;
 	ump->softdep_on_journal = 0;
 	ump->softdep_accdeps = 0;
 	ump->softdep_req = 0;
 	ump->softdep_jblocks = NULL;
 	error = softdep_journal_lookup(mp, &vp);
 	if (error != 0) {
 		printf("Failed to find journal.  Use tunefs to create one\n");
 		return (error);
 	}
 	ip = VTOI(vp);
 	if (ip->i_size < SUJ_MIN) {
 		error = ENOSPC;
 		goto out;
 	}
 	bcount = lblkno(fs, ip->i_size);	/* Only use whole blocks. */
 	jblocks = jblocks_create();
 	for (i = 0; i < bcount; i++) {
 		error = ufs_bmaparray(vp, i, &blkno, NULL, NULL, NULL);
 		if (error)
 			break;
 		jblocks_add(jblocks, blkno, fsbtodb(fs, fs->fs_frag));
 	}
 	if (error) {
 		jblocks_destroy(jblocks);
 		goto out;
 	}
 	jblocks->jb_low = jblocks->jb_free / 3;	/* Reserve 33%. */
 	jblocks->jb_min = jblocks->jb_free / 10; /* Suspend at 10%. */
 	ump->softdep_jblocks = jblocks;
 out:
 	if (error == 0) {
 		MNT_ILOCK(mp);
 		mp->mnt_flag |= MNT_SUJ;
 		mp->mnt_flag &= ~MNT_SOFTDEP;
 		MNT_IUNLOCK(mp);
 		/*
 		 * Only validate the journal contents if the
 		 * filesystem is clean, otherwise we write the logs
 		 * but they'll never be used.  If the filesystem was
 		 * still dirty when we mounted it the journal is
 		 * invalid and a new journal can only be valid if it
 		 * starts from a clean mount.
 		 */
 		if (fs->fs_clean) {
 			DIP_SET(ip, i_modrev, fs->fs_mtime);
 			ip->i_flags |= IN_MODIFIED;
 			ffs_update(vp, 1);
 		}
 	}
 	vput(vp);
 	return (error);
 }
 
 static void
 journal_unmount(ump)
 	struct ufsmount *ump;
 {
 
 	if (ump->softdep_jblocks)
 		jblocks_destroy(ump->softdep_jblocks);
 	ump->softdep_jblocks = NULL;
 }
 
 /*
  * Called when a journal record is ready to be written.  Space is allocated
  * and the journal entry is created when the journal is flushed to stable
  * store.
  */
 static void
 add_to_journal(wk)
 	struct worklist *wk;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(wk->wk_mp);
 	LOCK_OWNED(ump);
 	if (wk->wk_state & ONWORKLIST)
 		panic("add_to_journal: %s(0x%X) already on list",
 		    TYPENAME(wk->wk_type), wk->wk_state);
 	wk->wk_state |= ONWORKLIST | DEPCOMPLETE;
 	if (LIST_EMPTY(&ump->softdep_journal_pending)) {
 		ump->softdep_jblocks->jb_age = ticks;
 		LIST_INSERT_HEAD(&ump->softdep_journal_pending, wk, wk_list);
 	} else
 		LIST_INSERT_AFTER(ump->softdep_journal_tail, wk, wk_list);
 	ump->softdep_journal_tail = wk;
 	ump->softdep_on_journal += 1;
 }
 
 /*
  * Remove an arbitrary item for the journal worklist maintain the tail
  * pointer.  This happens when a new operation obviates the need to
  * journal an old operation.
  */
 static void
 remove_from_journal(wk)
 	struct worklist *wk;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(wk->wk_mp);
 	LOCK_OWNED(ump);
 #ifdef SUJ_DEBUG
 	{
 		struct worklist *wkn;
 
 		LIST_FOREACH(wkn, &ump->softdep_journal_pending, wk_list)
 			if (wkn == wk)
 				break;
 		if (wkn == NULL)
 			panic("remove_from_journal: %p is not in journal", wk);
 	}
 #endif
 	/*
 	 * We emulate a TAILQ to save space in most structures which do not
 	 * require TAILQ semantics.  Here we must update the tail position
 	 * when removing the tail which is not the final entry. This works
 	 * only if the worklist linkage are at the beginning of the structure.
 	 */
 	if (ump->softdep_journal_tail == wk)
 		ump->softdep_journal_tail =
 		    (struct worklist *)wk->wk_list.le_prev;
 
 	WORKLIST_REMOVE(wk);
 	ump->softdep_on_journal -= 1;
 }
 
 /*
  * Check for journal space as well as dependency limits so the prelink
  * code can throttle both journaled and non-journaled filesystems.
  * Threshold is 0 for low and 1 for min.
  */
 static int
 journal_space(ump, thresh)
 	struct ufsmount *ump;
 	int thresh;
 {
 	struct jblocks *jblocks;
 	int limit, avail;
 
 	jblocks = ump->softdep_jblocks;
 	if (jblocks == NULL)
 		return (1);
 	/*
 	 * We use a tighter restriction here to prevent request_cleanup()
 	 * running in threads from running into locks we currently hold.
 	 * We have to be over the limit and our filesystem has to be
 	 * responsible for more than our share of that usage.
 	 */
 	limit = (max_softdeps / 10) * 9;
 	if (dep_current[D_INODEDEP] > limit &&
 	    ump->softdep_curdeps[D_INODEDEP] > limit / stat_flush_threads)
 		return (0);
 	if (thresh)
 		thresh = jblocks->jb_min;
 	else
 		thresh = jblocks->jb_low;
 	avail = (ump->softdep_on_journal * JREC_SIZE) / DEV_BSIZE;
 	avail = jblocks->jb_free - avail;
 
 	return (avail > thresh);
 }
 
 static void
 journal_suspend(ump)
 	struct ufsmount *ump;
 {
 	struct jblocks *jblocks;
 	struct mount *mp;
 
 	mp = UFSTOVFS(ump);
 	jblocks = ump->softdep_jblocks;
 	MNT_ILOCK(mp);
 	if ((mp->mnt_kern_flag & MNTK_SUSPEND) == 0) {
 		stat_journal_min++;
 		mp->mnt_kern_flag |= MNTK_SUSPEND;
 		mp->mnt_susp_owner = ump->softdep_flushtd;
 	}
 	jblocks->jb_suspended = 1;
 	MNT_IUNLOCK(mp);
 }
 
 static int
 journal_unsuspend(struct ufsmount *ump)
 {
 	struct jblocks *jblocks;
 	struct mount *mp;
 
 	mp = UFSTOVFS(ump);
 	jblocks = ump->softdep_jblocks;
 
 	if (jblocks != NULL && jblocks->jb_suspended &&
 	    journal_space(ump, jblocks->jb_min)) {
 		jblocks->jb_suspended = 0;
 		FREE_LOCK(ump);
 		mp->mnt_susp_owner = curthread;
 		vfs_write_resume(mp, 0);
 		ACQUIRE_LOCK(ump);
 		return (1);
 	}
 	return (0);
 }
 
 /*
  * Called before any allocation function to be certain that there is
  * sufficient space in the journal prior to creating any new records.
  * Since in the case of block allocation we may have multiple locked
  * buffers at the time of the actual allocation we can not block
  * when the journal records are created.  Doing so would create a deadlock
  * if any of these buffers needed to be flushed to reclaim space.  Instead
  * we require a sufficiently large amount of available space such that
  * each thread in the system could have passed this allocation check and
  * still have sufficient free space.  With 20% of a minimum journal size
  * of 1MB we have 6553 records available.
  */
 int
 softdep_prealloc(vp, waitok)
 	struct vnode *vp;
 	int waitok;
 {
 	struct ufsmount *ump;
 
 	KASSERT(MOUNTEDSOFTDEP(vp->v_mount) != 0,
 	    ("softdep_prealloc called on non-softdep filesystem"));
 	/*
 	 * Nothing to do if we are not running journaled soft updates.
 	 * If we currently hold the snapshot lock, we must avoid handling
 	 * other resources that could cause deadlock.
 	 */
 	if (DOINGSUJ(vp) == 0 || IS_SNAPSHOT(VTOI(vp)))
 		return (0);
 	ump = VFSTOUFS(vp->v_mount);
 	ACQUIRE_LOCK(ump);
 	if (journal_space(ump, 0)) {
 		FREE_LOCK(ump);
 		return (0);
 	}
 	stat_journal_low++;
 	FREE_LOCK(ump);
 	if (waitok == MNT_NOWAIT)
 		return (ENOSPC);
 	/*
 	 * Attempt to sync this vnode once to flush any journal
 	 * work attached to it.
 	 */
 	if ((curthread->td_pflags & TDP_COWINPROGRESS) == 0)
 		ffs_syncvnode(vp, waitok, 0);
 	ACQUIRE_LOCK(ump);
 	process_removes(vp);
 	process_truncates(vp);
 	if (journal_space(ump, 0) == 0) {
 		softdep_speedup(ump);
 		if (journal_space(ump, 1) == 0)
 			journal_suspend(ump);
 	}
 	FREE_LOCK(ump);
 
 	return (0);
 }
 
 /*
  * Before adjusting a link count on a vnode verify that we have sufficient
  * journal space.  If not, process operations that depend on the currently
  * locked pair of vnodes to try to flush space as the syncer, buf daemon,
  * and softdep flush threads can not acquire these locks to reclaim space.
  */
 static void
 softdep_prelink(dvp, vp)
 	struct vnode *dvp;
 	struct vnode *vp;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(dvp->v_mount);
 	LOCK_OWNED(ump);
 	/*
 	 * Nothing to do if we have sufficient journal space.
 	 * If we currently hold the snapshot lock, we must avoid
 	 * handling other resources that could cause deadlock.
 	 */
 	if (journal_space(ump, 0) || (vp && IS_SNAPSHOT(VTOI(vp))))
 		return;
 	stat_journal_low++;
 	FREE_LOCK(ump);
 	if (vp)
 		ffs_syncvnode(vp, MNT_NOWAIT, 0);
 	ffs_syncvnode(dvp, MNT_WAIT, 0);
 	ACQUIRE_LOCK(ump);
 	/* Process vp before dvp as it may create .. removes. */
 	if (vp) {
 		process_removes(vp);
 		process_truncates(vp);
 	}
 	process_removes(dvp);
 	process_truncates(dvp);
 	softdep_speedup(ump);
 	process_worklist_item(UFSTOVFS(ump), 2, LK_NOWAIT);
 	if (journal_space(ump, 0) == 0) {
 		softdep_speedup(ump);
 		if (journal_space(ump, 1) == 0)
 			journal_suspend(ump);
 	}
 }
 
 static void
 jseg_write(ump, jseg, data)
 	struct ufsmount *ump;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jsegrec *rec;
 
 	rec = (struct jsegrec *)data;
 	rec->jsr_seq = jseg->js_seq;
 	rec->jsr_oldest = jseg->js_oldseq;
 	rec->jsr_cnt = jseg->js_cnt;
 	rec->jsr_blocks = jseg->js_size / ump->um_devvp->v_bufobj.bo_bsize;
 	rec->jsr_crc = 0;
 	rec->jsr_time = ump->um_fs->fs_mtime;
 }
 
 static inline void
 inoref_write(inoref, jseg, rec)
 	struct inoref *inoref;
 	struct jseg *jseg;
 	struct jrefrec *rec;
 {
 
 	inoref->if_jsegdep->jd_seg = jseg;
 	rec->jr_ino = inoref->if_ino;
 	rec->jr_parent = inoref->if_parent;
 	rec->jr_nlink = inoref->if_nlink;
 	rec->jr_mode = inoref->if_mode;
 	rec->jr_diroff = inoref->if_diroff;
 }
 
 static void
 jaddref_write(jaddref, jseg, data)
 	struct jaddref *jaddref;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jrefrec *rec;
 
 	rec = (struct jrefrec *)data;
 	rec->jr_op = JOP_ADDREF;
 	inoref_write(&jaddref->ja_ref, jseg, rec);
 }
 
 static void
 jremref_write(jremref, jseg, data)
 	struct jremref *jremref;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jrefrec *rec;
 
 	rec = (struct jrefrec *)data;
 	rec->jr_op = JOP_REMREF;
 	inoref_write(&jremref->jr_ref, jseg, rec);
 }
 
 static void
 jmvref_write(jmvref, jseg, data)
 	struct jmvref *jmvref;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jmvrec *rec;
 
 	rec = (struct jmvrec *)data;
 	rec->jm_op = JOP_MVREF;
 	rec->jm_ino = jmvref->jm_ino;
 	rec->jm_parent = jmvref->jm_parent;
 	rec->jm_oldoff = jmvref->jm_oldoff;
 	rec->jm_newoff = jmvref->jm_newoff;
 }
 
 static void
 jnewblk_write(jnewblk, jseg, data)
 	struct jnewblk *jnewblk;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jblkrec *rec;
 
 	jnewblk->jn_jsegdep->jd_seg = jseg;
 	rec = (struct jblkrec *)data;
 	rec->jb_op = JOP_NEWBLK;
 	rec->jb_ino = jnewblk->jn_ino;
 	rec->jb_blkno = jnewblk->jn_blkno;
 	rec->jb_lbn = jnewblk->jn_lbn;
 	rec->jb_frags = jnewblk->jn_frags;
 	rec->jb_oldfrags = jnewblk->jn_oldfrags;
 }
 
 static void
 jfreeblk_write(jfreeblk, jseg, data)
 	struct jfreeblk *jfreeblk;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jblkrec *rec;
 
 	jfreeblk->jf_dep.jb_jsegdep->jd_seg = jseg;
 	rec = (struct jblkrec *)data;
 	rec->jb_op = JOP_FREEBLK;
 	rec->jb_ino = jfreeblk->jf_ino;
 	rec->jb_blkno = jfreeblk->jf_blkno;
 	rec->jb_lbn = jfreeblk->jf_lbn;
 	rec->jb_frags = jfreeblk->jf_frags;
 	rec->jb_oldfrags = 0;
 }
 
 static void
 jfreefrag_write(jfreefrag, jseg, data)
 	struct jfreefrag *jfreefrag;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jblkrec *rec;
 
 	jfreefrag->fr_jsegdep->jd_seg = jseg;
 	rec = (struct jblkrec *)data;
 	rec->jb_op = JOP_FREEBLK;
 	rec->jb_ino = jfreefrag->fr_ino;
 	rec->jb_blkno = jfreefrag->fr_blkno;
 	rec->jb_lbn = jfreefrag->fr_lbn;
 	rec->jb_frags = jfreefrag->fr_frags;
 	rec->jb_oldfrags = 0;
 }
 
 static void
 jtrunc_write(jtrunc, jseg, data)
 	struct jtrunc *jtrunc;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jtrncrec *rec;
 
 	jtrunc->jt_dep.jb_jsegdep->jd_seg = jseg;
 	rec = (struct jtrncrec *)data;
 	rec->jt_op = JOP_TRUNC;
 	rec->jt_ino = jtrunc->jt_ino;
 	rec->jt_size = jtrunc->jt_size;
 	rec->jt_extsize = jtrunc->jt_extsize;
 }
 
 static void
 jfsync_write(jfsync, jseg, data)
 	struct jfsync *jfsync;
 	struct jseg *jseg;
 	uint8_t *data;
 {
 	struct jtrncrec *rec;
 
 	rec = (struct jtrncrec *)data;
 	rec->jt_op = JOP_SYNC;
 	rec->jt_ino = jfsync->jfs_ino;
 	rec->jt_size = jfsync->jfs_size;
 	rec->jt_extsize = jfsync->jfs_extsize;
 }
 
 static void
 softdep_flushjournal(mp)
 	struct mount *mp;
 {
 	struct jblocks *jblocks;
 	struct ufsmount *ump;
 
 	if (MOUNTEDSUJ(mp) == 0)
 		return;
 	ump = VFSTOUFS(mp);
 	jblocks = ump->softdep_jblocks;
 	ACQUIRE_LOCK(ump);
 	while (ump->softdep_on_journal) {
 		jblocks->jb_needseg = 1;
 		softdep_process_journal(mp, NULL, MNT_WAIT);
 	}
 	FREE_LOCK(ump);
 }
 
 static void softdep_synchronize_completed(struct bio *);
 static void softdep_synchronize(struct bio *, struct ufsmount *, void *);
 
 static void
 softdep_synchronize_completed(bp)
         struct bio *bp;
 {
 	struct jseg *oldest;
 	struct jseg *jseg;
 	struct ufsmount *ump;
 
 	/*
 	 * caller1 marks the last segment written before we issued the
 	 * synchronize cache.
 	 */
 	jseg = bp->bio_caller1;
 	if (jseg == NULL) {
 		g_destroy_bio(bp);
 		return;
 	}
 	ump = VFSTOUFS(jseg->js_list.wk_mp);
 	ACQUIRE_LOCK(ump);
 	oldest = NULL;
 	/*
 	 * Mark all the journal entries waiting on the synchronize cache
 	 * as completed so they may continue on.
 	 */
 	while (jseg != NULL && (jseg->js_state & COMPLETE) == 0) {
 		jseg->js_state |= COMPLETE;
 		oldest = jseg;
 		jseg = TAILQ_PREV(jseg, jseglst, js_next);
 	}
 	/*
 	 * Restart deferred journal entry processing from the oldest
 	 * completed jseg.
 	 */
 	if (oldest)
 		complete_jsegs(oldest);
 
 	FREE_LOCK(ump);
 	g_destroy_bio(bp);
 }
 
 /*
  * Send BIO_FLUSH/SYNCHRONIZE CACHE to the device to enforce write ordering
  * barriers.  The journal must be written prior to any blocks that depend
  * on it and the journal can not be released until the blocks have be
  * written.  This code handles both barriers simultaneously.
  */
 static void
 softdep_synchronize(bp, ump, caller1)
 	struct bio *bp;
 	struct ufsmount *ump;
 	void *caller1;
 {
 
 	bp->bio_cmd = BIO_FLUSH;
 	bp->bio_flags |= BIO_ORDERED;
 	bp->bio_data = NULL;
 	bp->bio_offset = ump->um_cp->provider->mediasize;
 	bp->bio_length = 0;
 	bp->bio_done = softdep_synchronize_completed;
 	bp->bio_caller1 = caller1;
 	g_io_request(bp,
 	    (struct g_consumer *)ump->um_devvp->v_bufobj.bo_private);
 }
 
 /*
  * Flush some journal records to disk.
  */
 static void
 softdep_process_journal(mp, needwk, flags)
 	struct mount *mp;
 	struct worklist *needwk;
 	int flags;
 {
 	struct jblocks *jblocks;
 	struct ufsmount *ump;
 	struct worklist *wk;
 	struct jseg *jseg;
 	struct buf *bp;
 	struct bio *bio;
 	uint8_t *data;
 	struct fs *fs;
 	int shouldflush;
 	int segwritten;
 	int jrecmin;	/* Minimum records per block. */
 	int jrecmax;	/* Maximum records per block. */
 	int size;
 	int cnt;
 	int off;
 	int devbsize;
 
 	if (MOUNTEDSUJ(mp) == 0)
 		return;
 	shouldflush = softdep_flushcache;
 	bio = NULL;
 	jseg = NULL;
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	fs = ump->um_fs;
 	jblocks = ump->softdep_jblocks;
 	devbsize = ump->um_devvp->v_bufobj.bo_bsize;
 	/*
 	 * We write anywhere between a disk block and fs block.  The upper
 	 * bound is picked to prevent buffer cache fragmentation and limit
 	 * processing time per I/O.
 	 */
 	jrecmin = (devbsize / JREC_SIZE) - 1; /* -1 for seg header */
 	jrecmax = (fs->fs_bsize / devbsize) * jrecmin;
 	segwritten = 0;
 	for (;;) {
 		cnt = ump->softdep_on_journal;
 		/*
 		 * Criteria for writing a segment:
 		 * 1) We have a full block.
 		 * 2) We're called from jwait() and haven't found the
 		 *    journal item yet.
 		 * 3) Always write if needseg is set.
 		 * 4) If we are called from process_worklist and have
 		 *    not yet written anything we write a partial block
 		 *    to enforce a 1 second maximum latency on journal
 		 *    entries.
 		 */
 		if (cnt < (jrecmax - 1) && needwk == NULL &&
 		    jblocks->jb_needseg == 0 && (segwritten || cnt == 0))
 			break;
 		cnt++;
 		/*
 		 * Verify some free journal space.  softdep_prealloc() should
 		 * guarantee that we don't run out so this is indicative of
 		 * a problem with the flow control.  Try to recover
 		 * gracefully in any event.
 		 */
 		while (jblocks->jb_free == 0) {
 			if (flags != MNT_WAIT)
 				break;
 			printf("softdep: Out of journal space!\n");
 			softdep_speedup(ump);
 			msleep(jblocks, LOCK_PTR(ump), PRIBIO, "jblocks", hz);
 		}
 		FREE_LOCK(ump);
 		jseg = malloc(sizeof(*jseg), M_JSEG, M_SOFTDEP_FLAGS);
 		workitem_alloc(&jseg->js_list, D_JSEG, mp);
 		LIST_INIT(&jseg->js_entries);
 		LIST_INIT(&jseg->js_indirs);
 		jseg->js_state = ATTACHED;
 		if (shouldflush == 0)
 			jseg->js_state |= COMPLETE;
 		else if (bio == NULL)
 			bio = g_alloc_bio();
 		jseg->js_jblocks = jblocks;
 		bp = geteblk(fs->fs_bsize, 0);
 		ACQUIRE_LOCK(ump);
 		/*
 		 * If there was a race while we were allocating the block
 		 * and jseg the entry we care about was likely written.
 		 * We bail out in both the WAIT and NOWAIT case and assume
 		 * the caller will loop if the entry it cares about is
 		 * not written.
 		 */
 		cnt = ump->softdep_on_journal;
 		if (cnt + jblocks->jb_needseg == 0 || jblocks->jb_free == 0) {
 			bp->b_flags |= B_INVAL | B_NOCACHE;
 			WORKITEM_FREE(jseg, D_JSEG);
 			FREE_LOCK(ump);
 			brelse(bp);
 			ACQUIRE_LOCK(ump);
 			break;
 		}
 		/*
 		 * Calculate the disk block size required for the available
 		 * records rounded to the min size.
 		 */
 		if (cnt == 0)
 			size = devbsize;
 		else if (cnt < jrecmax)
 			size = howmany(cnt, jrecmin) * devbsize;
 		else
 			size = fs->fs_bsize;
 		/*
 		 * Allocate a disk block for this journal data and account
 		 * for truncation of the requested size if enough contiguous
 		 * space was not available.
 		 */
 		bp->b_blkno = jblocks_alloc(jblocks, size, &size);
 		bp->b_lblkno = bp->b_blkno;
 		bp->b_offset = bp->b_blkno * DEV_BSIZE;
 		bp->b_bcount = size;
 		bp->b_flags &= ~B_INVAL;
 		bp->b_flags |= B_VALIDSUSPWRT | B_NOCOPY;
 		/*
 		 * Initialize our jseg with cnt records.  Assign the next
 		 * sequence number to it and link it in-order.
 		 */
 		cnt = MIN(cnt, (size / devbsize) * jrecmin);
 		jseg->js_buf = bp;
 		jseg->js_cnt = cnt;
 		jseg->js_refs = cnt + 1;	/* Self ref. */
 		jseg->js_size = size;
 		jseg->js_seq = jblocks->jb_nextseq++;
 		if (jblocks->jb_oldestseg == NULL)
 			jblocks->jb_oldestseg = jseg;
 		jseg->js_oldseq = jblocks->jb_oldestseg->js_seq;
 		TAILQ_INSERT_TAIL(&jblocks->jb_segs, jseg, js_next);
 		if (jblocks->jb_writeseg == NULL)
 			jblocks->jb_writeseg = jseg;
 		/*
 		 * Start filling in records from the pending list.
 		 */
 		data = bp->b_data;
 		off = 0;
 
 		/*
 		 * Always put a header on the first block.
 		 * XXX As with below, there might not be a chance to get
 		 * into the loop.  Ensure that something valid is written.
 		 */
 		jseg_write(ump, jseg, data);
 		off += JREC_SIZE;
 		data = bp->b_data + off;
 
 		/*
 		 * XXX Something is wrong here.  There's no work to do,
 		 * but we need to perform and I/O and allow it to complete
 		 * anyways.
 		 */
 		if (LIST_EMPTY(&ump->softdep_journal_pending))
 			stat_emptyjblocks++;
 
 		while ((wk = LIST_FIRST(&ump->softdep_journal_pending))
 		    != NULL) {
 			if (cnt == 0)
 				break;
 			/* Place a segment header on every device block. */
 			if ((off % devbsize) == 0) {
 				jseg_write(ump, jseg, data);
 				off += JREC_SIZE;
 				data = bp->b_data + off;
 			}
 			if (wk == needwk)
 				needwk = NULL;
 			remove_from_journal(wk);
 			wk->wk_state |= INPROGRESS;
 			WORKLIST_INSERT(&jseg->js_entries, wk);
 			switch (wk->wk_type) {
 			case D_JADDREF:
 				jaddref_write(WK_JADDREF(wk), jseg, data);
 				break;
 			case D_JREMREF:
 				jremref_write(WK_JREMREF(wk), jseg, data);
 				break;
 			case D_JMVREF:
 				jmvref_write(WK_JMVREF(wk), jseg, data);
 				break;
 			case D_JNEWBLK:
 				jnewblk_write(WK_JNEWBLK(wk), jseg, data);
 				break;
 			case D_JFREEBLK:
 				jfreeblk_write(WK_JFREEBLK(wk), jseg, data);
 				break;
 			case D_JFREEFRAG:
 				jfreefrag_write(WK_JFREEFRAG(wk), jseg, data);
 				break;
 			case D_JTRUNC:
 				jtrunc_write(WK_JTRUNC(wk), jseg, data);
 				break;
 			case D_JFSYNC:
 				jfsync_write(WK_JFSYNC(wk), jseg, data);
 				break;
 			default:
 				panic("process_journal: Unknown type %s",
 				    TYPENAME(wk->wk_type));
 				/* NOTREACHED */
 			}
 			off += JREC_SIZE;
 			data = bp->b_data + off;
 			cnt--;
 		}
 
 		/* Clear any remaining space so we don't leak kernel data */
 		if (size > off)
 			bzero(data, size - off);
 
 		/*
 		 * Write this one buffer and continue.
 		 */
 		segwritten = 1;
 		jblocks->jb_needseg = 0;
 		WORKLIST_INSERT(&bp->b_dep, &jseg->js_list);
 		FREE_LOCK(ump);
 		pbgetvp(ump->um_devvp, bp);
 		/*
 		 * We only do the blocking wait once we find the journal
 		 * entry we're looking for.
 		 */
 		if (needwk == NULL && flags == MNT_WAIT)
 			bwrite(bp);
 		else
 			bawrite(bp);
 		ACQUIRE_LOCK(ump);
 	}
 	/*
 	 * If we wrote a segment issue a synchronize cache so the journal
 	 * is reflected on disk before the data is written.  Since reclaiming
 	 * journal space also requires writing a journal record this
 	 * process also enforces a barrier before reclamation.
 	 */
 	if (segwritten && shouldflush) {
 		softdep_synchronize(bio, ump, 
 		    TAILQ_LAST(&jblocks->jb_segs, jseglst));
 	} else if (bio)
 		g_destroy_bio(bio);
 	/*
 	 * If we've suspended the filesystem because we ran out of journal
 	 * space either try to sync it here to make some progress or
 	 * unsuspend it if we already have.
 	 */
 	if (flags == 0 && jblocks->jb_suspended) {
 		if (journal_unsuspend(ump))
 			return;
 		FREE_LOCK(ump);
 		VFS_SYNC(mp, MNT_NOWAIT);
 		ffs_sbupdate(ump, MNT_WAIT, 0);
 		ACQUIRE_LOCK(ump);
 	}
 }
 
 /*
  * Complete a jseg, allowing all dependencies awaiting journal writes
  * to proceed.  Each journal dependency also attaches a jsegdep to dependent
  * structures so that the journal segment can be freed to reclaim space.
  */
 static void
 complete_jseg(jseg)
 	struct jseg *jseg;
 {
 	struct worklist *wk;
 	struct jmvref *jmvref;
 	int waiting;
 #ifdef INVARIANTS
 	int i = 0;
 #endif
 
 	while ((wk = LIST_FIRST(&jseg->js_entries)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		waiting = wk->wk_state & IOWAITING;
 		wk->wk_state &= ~(INPROGRESS | IOWAITING);
 		wk->wk_state |= COMPLETE;
 		KASSERT(i++ < jseg->js_cnt,
 		    ("handle_written_jseg: overflow %d >= %d",
 		    i - 1, jseg->js_cnt));
 		switch (wk->wk_type) {
 		case D_JADDREF:
 			handle_written_jaddref(WK_JADDREF(wk));
 			break;
 		case D_JREMREF:
 			handle_written_jremref(WK_JREMREF(wk));
 			break;
 		case D_JMVREF:
 			rele_jseg(jseg);	/* No jsegdep. */
 			jmvref = WK_JMVREF(wk);
 			LIST_REMOVE(jmvref, jm_deps);
 			if ((jmvref->jm_pagedep->pd_state & ONWORKLIST) == 0)
 				free_pagedep(jmvref->jm_pagedep);
 			WORKITEM_FREE(jmvref, D_JMVREF);
 			break;
 		case D_JNEWBLK:
 			handle_written_jnewblk(WK_JNEWBLK(wk));
 			break;
 		case D_JFREEBLK:
 			handle_written_jblkdep(&WK_JFREEBLK(wk)->jf_dep);
 			break;
 		case D_JTRUNC:
 			handle_written_jblkdep(&WK_JTRUNC(wk)->jt_dep);
 			break;
 		case D_JFSYNC:
 			rele_jseg(jseg);	/* No jsegdep. */
 			WORKITEM_FREE(wk, D_JFSYNC);
 			break;
 		case D_JFREEFRAG:
 			handle_written_jfreefrag(WK_JFREEFRAG(wk));
 			break;
 		default:
 			panic("handle_written_jseg: Unknown type %s",
 			    TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 		if (waiting)
 			wakeup(wk);
 	}
 	/* Release the self reference so the structure may be freed. */
 	rele_jseg(jseg);
 }
 
 /*
  * Determine which jsegs are ready for completion processing.  Waits for
  * synchronize cache to complete as well as forcing in-order completion
  * of journal entries.
  */
 static void
 complete_jsegs(jseg)
 	struct jseg *jseg;
 {
 	struct jblocks *jblocks;
 	struct jseg *jsegn;
 
 	jblocks = jseg->js_jblocks;
 	/*
 	 * Don't allow out of order completions.  If this isn't the first
 	 * block wait for it to write before we're done.
 	 */
 	if (jseg != jblocks->jb_writeseg)
 		return;
 	/* Iterate through available jsegs processing their entries. */
 	while (jseg && (jseg->js_state & ALLCOMPLETE) == ALLCOMPLETE) {
 		jblocks->jb_oldestwrseq = jseg->js_oldseq;
 		jsegn = TAILQ_NEXT(jseg, js_next);
 		complete_jseg(jseg);
 		jseg = jsegn;
 	}
 	jblocks->jb_writeseg = jseg;
 	/*
 	 * Attempt to free jsegs now that oldestwrseq may have advanced. 
 	 */
 	free_jsegs(jblocks);
 }
 
 /*
  * Mark a jseg as DEPCOMPLETE and throw away the buffer.  Attempt to handle
  * the final completions.
  */
 static void
 handle_written_jseg(jseg, bp)
 	struct jseg *jseg;
 	struct buf *bp;
 {
 
 	if (jseg->js_refs == 0)
 		panic("handle_written_jseg: No self-reference on %p", jseg);
 	jseg->js_state |= DEPCOMPLETE;
 	/*
 	 * We'll never need this buffer again, set flags so it will be
 	 * discarded.
 	 */
 	bp->b_flags |= B_INVAL | B_NOCACHE;
 	pbrelvp(bp);
 	complete_jsegs(jseg);
 }
 
 static inline struct jsegdep *
 inoref_jseg(inoref)
 	struct inoref *inoref;
 {
 	struct jsegdep *jsegdep;
 
 	jsegdep = inoref->if_jsegdep;
 	inoref->if_jsegdep = NULL;
 
 	return (jsegdep);
 }
 
 /*
  * Called once a jremref has made it to stable store.  The jremref is marked
  * complete and we attempt to free it.  Any pagedeps writes sleeping waiting
  * for the jremref to complete will be awoken by free_jremref.
  */
 static void
 handle_written_jremref(jremref)
 	struct jremref *jremref;
 {
 	struct inodedep *inodedep;
 	struct jsegdep *jsegdep;
 	struct dirrem *dirrem;
 
 	/* Grab the jsegdep. */
 	jsegdep = inoref_jseg(&jremref->jr_ref);
 	/*
 	 * Remove us from the inoref list.
 	 */
 	if (inodedep_lookup(jremref->jr_list.wk_mp, jremref->jr_ref.if_ino,
 	    0, &inodedep) == 0)
 		panic("handle_written_jremref: Lost inodedep");
 	TAILQ_REMOVE(&inodedep->id_inoreflst, &jremref->jr_ref, if_deps);
 	/*
 	 * Complete the dirrem.
 	 */
 	dirrem = jremref->jr_dirrem;
 	jremref->jr_dirrem = NULL;
 	LIST_REMOVE(jremref, jr_deps);
 	jsegdep->jd_state |= jremref->jr_state & MKDIR_PARENT;
 	jwork_insert(&dirrem->dm_jwork, jsegdep);
 	if (LIST_EMPTY(&dirrem->dm_jremrefhd) &&
 	    (dirrem->dm_state & COMPLETE) != 0)
 		add_to_worklist(&dirrem->dm_list, 0);
 	free_jremref(jremref);
 }
 
 /*
  * Called once a jaddref has made it to stable store.  The dependency is
  * marked complete and any dependent structures are added to the inode
  * bufwait list to be completed as soon as it is written.  If a bitmap write
  * depends on this entry we move the inode into the inodedephd of the
  * bmsafemap dependency and attempt to remove the jaddref from the bmsafemap.
  */
 static void
 handle_written_jaddref(jaddref)
 	struct jaddref *jaddref;
 {
 	struct jsegdep *jsegdep;
 	struct inodedep *inodedep;
 	struct diradd *diradd;
 	struct mkdir *mkdir;
 
 	/* Grab the jsegdep. */
 	jsegdep = inoref_jseg(&jaddref->ja_ref);
 	mkdir = NULL;
 	diradd = NULL;
 	if (inodedep_lookup(jaddref->ja_list.wk_mp, jaddref->ja_ino,
 	    0, &inodedep) == 0)
 		panic("handle_written_jaddref: Lost inodedep.");
 	if (jaddref->ja_diradd == NULL)
 		panic("handle_written_jaddref: No dependency");
 	if (jaddref->ja_diradd->da_list.wk_type == D_DIRADD) {
 		diradd = jaddref->ja_diradd;
 		WORKLIST_INSERT(&inodedep->id_bufwait, &diradd->da_list);
 	} else if (jaddref->ja_state & MKDIR_PARENT) {
 		mkdir = jaddref->ja_mkdir;
 		WORKLIST_INSERT(&inodedep->id_bufwait, &mkdir->md_list);
 	} else if (jaddref->ja_state & MKDIR_BODY)
 		mkdir = jaddref->ja_mkdir;
 	else
 		panic("handle_written_jaddref: Unknown dependency %p",
 		    jaddref->ja_diradd);
 	jaddref->ja_diradd = NULL;	/* also clears ja_mkdir */
 	/*
 	 * Remove us from the inode list.
 	 */
 	TAILQ_REMOVE(&inodedep->id_inoreflst, &jaddref->ja_ref, if_deps);
 	/*
 	 * The mkdir may be waiting on the jaddref to clear before freeing.
 	 */
 	if (mkdir) {
 		KASSERT(mkdir->md_list.wk_type == D_MKDIR,
 		    ("handle_written_jaddref: Incorrect type for mkdir %s",
 		    TYPENAME(mkdir->md_list.wk_type)));
 		mkdir->md_jaddref = NULL;
 		diradd = mkdir->md_diradd;
 		mkdir->md_state |= DEPCOMPLETE;
 		complete_mkdir(mkdir);
 	}
 	jwork_insert(&diradd->da_jwork, jsegdep);
 	if (jaddref->ja_state & NEWBLOCK) {
 		inodedep->id_state |= ONDEPLIST;
 		LIST_INSERT_HEAD(&inodedep->id_bmsafemap->sm_inodedephd,
 		    inodedep, id_deps);
 	}
 	free_jaddref(jaddref);
 }
 
 /*
  * Called once a jnewblk journal is written.  The allocdirect or allocindir
  * is placed in the bmsafemap to await notification of a written bitmap.  If
  * the operation was canceled we add the segdep to the appropriate
  * dependency to free the journal space once the canceling operation
  * completes.
  */
 static void
 handle_written_jnewblk(jnewblk)
 	struct jnewblk *jnewblk;
 {
 	struct bmsafemap *bmsafemap;
 	struct freefrag *freefrag;
 	struct freework *freework;
 	struct jsegdep *jsegdep;
 	struct newblk *newblk;
 
 	/* Grab the jsegdep. */
 	jsegdep = jnewblk->jn_jsegdep;
 	jnewblk->jn_jsegdep = NULL;
 	if (jnewblk->jn_dep == NULL) 
 		panic("handle_written_jnewblk: No dependency for the segdep.");
 	switch (jnewblk->jn_dep->wk_type) {
 	case D_NEWBLK:
 	case D_ALLOCDIRECT:
 	case D_ALLOCINDIR:
 		/*
 		 * Add the written block to the bmsafemap so it can
 		 * be notified when the bitmap is on disk.
 		 */
 		newblk = WK_NEWBLK(jnewblk->jn_dep);
 		newblk->nb_jnewblk = NULL;
 		if ((newblk->nb_state & GOINGAWAY) == 0) {
 			bmsafemap = newblk->nb_bmsafemap;
 			newblk->nb_state |= ONDEPLIST;
 			LIST_INSERT_HEAD(&bmsafemap->sm_newblkhd, newblk,
 			    nb_deps);
 		}
 		jwork_insert(&newblk->nb_jwork, jsegdep);
 		break;
 	case D_FREEFRAG:
 		/*
 		 * A newblock being removed by a freefrag when replaced by
 		 * frag extension.
 		 */
 		freefrag = WK_FREEFRAG(jnewblk->jn_dep);
 		freefrag->ff_jdep = NULL;
 		jwork_insert(&freefrag->ff_jwork, jsegdep);
 		break;
 	case D_FREEWORK:
 		/*
 		 * A direct block was removed by truncate.
 		 */
 		freework = WK_FREEWORK(jnewblk->jn_dep);
 		freework->fw_jnewblk = NULL;
 		jwork_insert(&freework->fw_freeblks->fb_jwork, jsegdep);
 		break;
 	default:
 		panic("handle_written_jnewblk: Unknown type %d.",
 		    jnewblk->jn_dep->wk_type);
 	}
 	jnewblk->jn_dep = NULL;
 	free_jnewblk(jnewblk);
 }
 
 /*
  * Cancel a jfreefrag that won't be needed, probably due to colliding with
  * an in-flight allocation that has not yet been committed.  Divorce us
  * from the freefrag and mark it DEPCOMPLETE so that it may be added
  * to the worklist.
  */
 static void
 cancel_jfreefrag(jfreefrag)
 	struct jfreefrag *jfreefrag;
 {
 	struct freefrag *freefrag;
 
 	if (jfreefrag->fr_jsegdep) {
 		free_jsegdep(jfreefrag->fr_jsegdep);
 		jfreefrag->fr_jsegdep = NULL;
 	}
 	freefrag = jfreefrag->fr_freefrag;
 	jfreefrag->fr_freefrag = NULL;
 	free_jfreefrag(jfreefrag);
 	freefrag->ff_state |= DEPCOMPLETE;
 	CTR1(KTR_SUJ, "cancel_jfreefrag: blkno %jd", freefrag->ff_blkno);
 }
 
 /*
  * Free a jfreefrag when the parent freefrag is rendered obsolete.
  */
 static void
 free_jfreefrag(jfreefrag)
 	struct jfreefrag *jfreefrag;
 {
 
 	if (jfreefrag->fr_state & INPROGRESS)
 		WORKLIST_REMOVE(&jfreefrag->fr_list);
 	else if (jfreefrag->fr_state & ONWORKLIST)
 		remove_from_journal(&jfreefrag->fr_list);
 	if (jfreefrag->fr_freefrag != NULL)
 		panic("free_jfreefrag:  Still attached to a freefrag.");
 	WORKITEM_FREE(jfreefrag, D_JFREEFRAG);
 }
 
 /*
  * Called when the journal write for a jfreefrag completes.  The parent
  * freefrag is added to the worklist if this completes its dependencies.
  */
 static void
 handle_written_jfreefrag(jfreefrag)
 	struct jfreefrag *jfreefrag;
 {
 	struct jsegdep *jsegdep;
 	struct freefrag *freefrag;
 
 	/* Grab the jsegdep. */
 	jsegdep = jfreefrag->fr_jsegdep;
 	jfreefrag->fr_jsegdep = NULL;
 	freefrag = jfreefrag->fr_freefrag;
 	if (freefrag == NULL)
 		panic("handle_written_jfreefrag: No freefrag.");
 	freefrag->ff_state |= DEPCOMPLETE;
 	freefrag->ff_jdep = NULL;
 	jwork_insert(&freefrag->ff_jwork, jsegdep);
 	if ((freefrag->ff_state & ALLCOMPLETE) == ALLCOMPLETE)
 		add_to_worklist(&freefrag->ff_list, 0);
 	jfreefrag->fr_freefrag = NULL;
 	free_jfreefrag(jfreefrag);
 }
 
 /*
  * Called when the journal write for a jfreeblk completes.  The jfreeblk
  * is removed from the freeblks list of pending journal writes and the
  * jsegdep is moved to the freeblks jwork to be completed when all blocks
  * have been reclaimed.
  */
 static void
 handle_written_jblkdep(jblkdep)
 	struct jblkdep *jblkdep;
 {
 	struct freeblks *freeblks;
 	struct jsegdep *jsegdep;
 
 	/* Grab the jsegdep. */
 	jsegdep = jblkdep->jb_jsegdep;
 	jblkdep->jb_jsegdep = NULL;
 	freeblks = jblkdep->jb_freeblks;
 	LIST_REMOVE(jblkdep, jb_deps);
 	jwork_insert(&freeblks->fb_jwork, jsegdep);
 	/*
 	 * If the freeblks is all journaled, we can add it to the worklist.
 	 */
 	if (LIST_EMPTY(&freeblks->fb_jblkdephd) &&
 	    (freeblks->fb_state & ALLCOMPLETE) == ALLCOMPLETE)
 		add_to_worklist(&freeblks->fb_list, WK_NODELAY);
 
 	free_jblkdep(jblkdep);
 }
 
 static struct jsegdep *
 newjsegdep(struct worklist *wk)
 {
 	struct jsegdep *jsegdep;
 
 	jsegdep = malloc(sizeof(*jsegdep), M_JSEGDEP, M_SOFTDEP_FLAGS);
 	workitem_alloc(&jsegdep->jd_list, D_JSEGDEP, wk->wk_mp);
 	jsegdep->jd_seg = NULL;
 
 	return (jsegdep);
 }
 
 static struct jmvref *
 newjmvref(dp, ino, oldoff, newoff)
 	struct inode *dp;
 	ino_t ino;
 	off_t oldoff;
 	off_t newoff;
 {
 	struct jmvref *jmvref;
 
 	jmvref = malloc(sizeof(*jmvref), M_JMVREF, M_SOFTDEP_FLAGS);
 	workitem_alloc(&jmvref->jm_list, D_JMVREF, UFSTOVFS(dp->i_ump));
 	jmvref->jm_list.wk_state = ATTACHED | DEPCOMPLETE;
 	jmvref->jm_parent = dp->i_number;
 	jmvref->jm_ino = ino;
 	jmvref->jm_oldoff = oldoff;
 	jmvref->jm_newoff = newoff;
 
 	return (jmvref);
 }
 
 /*
  * Allocate a new jremref that tracks the removal of ip from dp with the
  * directory entry offset of diroff.  Mark the entry as ATTACHED and
  * DEPCOMPLETE as we have all the information required for the journal write
  * and the directory has already been removed from the buffer.  The caller
  * is responsible for linking the jremref into the pagedep and adding it
  * to the journal to write.  The MKDIR_PARENT flag is set if we're doing
  * a DOTDOT addition so handle_workitem_remove() can properly assign
  * the jsegdep when we're done.
  */
 static struct jremref *
 newjremref(struct dirrem *dirrem, struct inode *dp, struct inode *ip,
     off_t diroff, nlink_t nlink)
 {
 	struct jremref *jremref;
 
 	jremref = malloc(sizeof(*jremref), M_JREMREF, M_SOFTDEP_FLAGS);
 	workitem_alloc(&jremref->jr_list, D_JREMREF, UFSTOVFS(dp->i_ump));
 	jremref->jr_state = ATTACHED;
 	newinoref(&jremref->jr_ref, ip->i_number, dp->i_number, diroff,
 	   nlink, ip->i_mode);
 	jremref->jr_dirrem = dirrem;
 
 	return (jremref);
 }
 
 static inline void
 newinoref(struct inoref *inoref, ino_t ino, ino_t parent, off_t diroff,
     nlink_t nlink, uint16_t mode)
 {
 
 	inoref->if_jsegdep = newjsegdep(&inoref->if_list);
 	inoref->if_diroff = diroff;
 	inoref->if_ino = ino;
 	inoref->if_parent = parent;
 	inoref->if_nlink = nlink;
 	inoref->if_mode = mode;
 }
 
 /*
  * Allocate a new jaddref to track the addition of ino to dp at diroff.  The
  * directory offset may not be known until later.  The caller is responsible
  * adding the entry to the journal when this information is available.  nlink
  * should be the link count prior to the addition and mode is only required
  * to have the correct FMT.
  */
 static struct jaddref *
 newjaddref(struct inode *dp, ino_t ino, off_t diroff, int16_t nlink,
     uint16_t mode)
 {
 	struct jaddref *jaddref;
 
 	jaddref = malloc(sizeof(*jaddref), M_JADDREF, M_SOFTDEP_FLAGS);
 	workitem_alloc(&jaddref->ja_list, D_JADDREF, UFSTOVFS(dp->i_ump));
 	jaddref->ja_state = ATTACHED;
 	jaddref->ja_mkdir = NULL;
 	newinoref(&jaddref->ja_ref, ino, dp->i_number, diroff, nlink, mode);
 
 	return (jaddref);
 }
 
 /*
  * Create a new free dependency for a freework.  The caller is responsible
  * for adjusting the reference count when it has the lock held.  The freedep
  * will track an outstanding bitmap write that will ultimately clear the
  * freework to continue.
  */
 static struct freedep *
 newfreedep(struct freework *freework)
 {
 	struct freedep *freedep;
 
 	freedep = malloc(sizeof(*freedep), M_FREEDEP, M_SOFTDEP_FLAGS);
 	workitem_alloc(&freedep->fd_list, D_FREEDEP, freework->fw_list.wk_mp);
 	freedep->fd_freework = freework;
 
 	return (freedep);
 }
 
 /*
  * Free a freedep structure once the buffer it is linked to is written.  If
  * this is the last reference to the freework schedule it for completion.
  */
 static void
 free_freedep(freedep)
 	struct freedep *freedep;
 {
 	struct freework *freework;
 
 	freework = freedep->fd_freework;
 	freework->fw_freeblks->fb_cgwait--;
 	if (--freework->fw_ref == 0)
 		freework_enqueue(freework);
 	WORKITEM_FREE(freedep, D_FREEDEP);
 }
 
 /*
  * Allocate a new freework structure that may be a level in an indirect
  * when parent is not NULL or a top level block when it is.  The top level
  * freework structures are allocated without the per-filesystem lock held
  * and before the freeblks is visible outside of softdep_setup_freeblocks().
  */
 static struct freework *
 newfreework(ump, freeblks, parent, lbn, nb, frags, off, journal)
 	struct ufsmount *ump;
 	struct freeblks *freeblks;
 	struct freework *parent;
 	ufs_lbn_t lbn;
 	ufs2_daddr_t nb;
 	int frags;
 	int off;
 	int journal;
 {
 	struct freework *freework;
 
 	freework = malloc(sizeof(*freework), M_FREEWORK, M_SOFTDEP_FLAGS);
 	workitem_alloc(&freework->fw_list, D_FREEWORK, freeblks->fb_list.wk_mp);
 	freework->fw_state = ATTACHED;
 	freework->fw_jnewblk = NULL;
 	freework->fw_freeblks = freeblks;
 	freework->fw_parent = parent;
 	freework->fw_lbn = lbn;
 	freework->fw_blkno = nb;
 	freework->fw_frags = frags;
 	freework->fw_indir = NULL;
 	freework->fw_ref = (MOUNTEDSUJ(UFSTOVFS(ump)) == 0 || lbn >= -NXADDR)
 		? 0 : NINDIR(ump->um_fs) + 1;
 	freework->fw_start = freework->fw_off = off;
 	if (journal)
 		newjfreeblk(freeblks, lbn, nb, frags);
 	if (parent == NULL) {
 		ACQUIRE_LOCK(ump);
 		WORKLIST_INSERT(&freeblks->fb_freeworkhd, &freework->fw_list);
 		freeblks->fb_ref++;
 		FREE_LOCK(ump);
 	}
 
 	return (freework);
 }
 
 /*
  * Eliminate a jfreeblk for a block that does not need journaling.
  */
 static void
 cancel_jfreeblk(freeblks, blkno)
 	struct freeblks *freeblks;
 	ufs2_daddr_t blkno;
 {
 	struct jfreeblk *jfreeblk;
 	struct jblkdep *jblkdep;
 
 	LIST_FOREACH(jblkdep, &freeblks->fb_jblkdephd, jb_deps) {
 		if (jblkdep->jb_list.wk_type != D_JFREEBLK)
 			continue;
 		jfreeblk = WK_JFREEBLK(&jblkdep->jb_list);
 		if (jfreeblk->jf_blkno == blkno)
 			break;
 	}
 	if (jblkdep == NULL)
 		return;
 	CTR1(KTR_SUJ, "cancel_jfreeblk: blkno %jd", blkno);
 	free_jsegdep(jblkdep->jb_jsegdep);
 	LIST_REMOVE(jblkdep, jb_deps);
 	WORKITEM_FREE(jfreeblk, D_JFREEBLK);
 }
 
 /*
  * Allocate a new jfreeblk to journal top level block pointer when truncating
  * a file.  The caller must add this to the worklist when the per-filesystem
  * lock is held.
  */
 static struct jfreeblk *
 newjfreeblk(freeblks, lbn, blkno, frags)
 	struct freeblks *freeblks;
 	ufs_lbn_t lbn;
 	ufs2_daddr_t blkno;
 	int frags;
 {
 	struct jfreeblk *jfreeblk;
 
 	jfreeblk = malloc(sizeof(*jfreeblk), M_JFREEBLK, M_SOFTDEP_FLAGS);
 	workitem_alloc(&jfreeblk->jf_dep.jb_list, D_JFREEBLK,
 	    freeblks->fb_list.wk_mp);
 	jfreeblk->jf_dep.jb_jsegdep = newjsegdep(&jfreeblk->jf_dep.jb_list);
 	jfreeblk->jf_dep.jb_freeblks = freeblks;
 	jfreeblk->jf_ino = freeblks->fb_inum;
 	jfreeblk->jf_lbn = lbn;
 	jfreeblk->jf_blkno = blkno;
 	jfreeblk->jf_frags = frags;
 	LIST_INSERT_HEAD(&freeblks->fb_jblkdephd, &jfreeblk->jf_dep, jb_deps);
 
 	return (jfreeblk);
 }
 
 /*
  * The journal is only prepared to handle full-size block numbers, so we
  * have to adjust the record to reflect the change to a full-size block.
  * For example, suppose we have a block made up of fragments 8-15 and
  * want to free its last two fragments. We are given a request that says:
  *     FREEBLK ino=5, blkno=14, lbn=0, frags=2, oldfrags=0
  * where frags are the number of fragments to free and oldfrags are the
  * number of fragments to keep. To block align it, we have to change it to
  * have a valid full-size blkno, so it becomes:
  *     FREEBLK ino=5, blkno=8, lbn=0, frags=2, oldfrags=6
  */
 static void
 adjust_newfreework(freeblks, frag_offset)
 	struct freeblks *freeblks;
 	int frag_offset;
 {
 	struct jfreeblk *jfreeblk;
 
 	KASSERT((LIST_FIRST(&freeblks->fb_jblkdephd) != NULL &&
 	    LIST_FIRST(&freeblks->fb_jblkdephd)->jb_list.wk_type == D_JFREEBLK),
 	    ("adjust_newfreework: Missing freeblks dependency"));
 
 	jfreeblk = WK_JFREEBLK(LIST_FIRST(&freeblks->fb_jblkdephd));
 	jfreeblk->jf_blkno -= frag_offset;
 	jfreeblk->jf_frags += frag_offset;
 }
 
 /*
  * Allocate a new jtrunc to track a partial truncation.
  */
 static struct jtrunc *
 newjtrunc(freeblks, size, extsize)
 	struct freeblks *freeblks;
 	off_t size;
 	int extsize;
 {
 	struct jtrunc *jtrunc;
 
 	jtrunc = malloc(sizeof(*jtrunc), M_JTRUNC, M_SOFTDEP_FLAGS);
 	workitem_alloc(&jtrunc->jt_dep.jb_list, D_JTRUNC,
 	    freeblks->fb_list.wk_mp);
 	jtrunc->jt_dep.jb_jsegdep = newjsegdep(&jtrunc->jt_dep.jb_list);
 	jtrunc->jt_dep.jb_freeblks = freeblks;
 	jtrunc->jt_ino = freeblks->fb_inum;
 	jtrunc->jt_size = size;
 	jtrunc->jt_extsize = extsize;
 	LIST_INSERT_HEAD(&freeblks->fb_jblkdephd, &jtrunc->jt_dep, jb_deps);
 
 	return (jtrunc);
 }
 
 /*
  * If we're canceling a new bitmap we have to search for another ref
  * to move into the bmsafemap dep.  This might be better expressed
  * with another structure.
  */
 static void
 move_newblock_dep(jaddref, inodedep)
 	struct jaddref *jaddref;
 	struct inodedep *inodedep;
 {
 	struct inoref *inoref;
 	struct jaddref *jaddrefn;
 
 	jaddrefn = NULL;
 	for (inoref = TAILQ_NEXT(&jaddref->ja_ref, if_deps); inoref;
 	    inoref = TAILQ_NEXT(inoref, if_deps)) {
 		if ((jaddref->ja_state & NEWBLOCK) &&
 		    inoref->if_list.wk_type == D_JADDREF) {
 			jaddrefn = (struct jaddref *)inoref;
 			break;
 		}
 	}
 	if (jaddrefn == NULL)
 		return;
 	jaddrefn->ja_state &= ~(ATTACHED | UNDONE);
 	jaddrefn->ja_state |= jaddref->ja_state &
 	    (ATTACHED | UNDONE | NEWBLOCK);
 	jaddref->ja_state &= ~(ATTACHED | UNDONE | NEWBLOCK);
 	jaddref->ja_state |= ATTACHED;
 	LIST_REMOVE(jaddref, ja_bmdeps);
 	LIST_INSERT_HEAD(&inodedep->id_bmsafemap->sm_jaddrefhd, jaddrefn,
 	    ja_bmdeps);
 }
 
 /*
  * Cancel a jaddref either before it has been written or while it is being
  * written.  This happens when a link is removed before the add reaches
  * the disk.  The jaddref dependency is kept linked into the bmsafemap
  * and inode to prevent the link count or bitmap from reaching the disk
  * until handle_workitem_remove() re-adjusts the counts and bitmaps as
  * required.
  *
  * Returns 1 if the canceled addref requires journaling of the remove and
  * 0 otherwise.
  */
 static int
 cancel_jaddref(jaddref, inodedep, wkhd)
 	struct jaddref *jaddref;
 	struct inodedep *inodedep;
 	struct workhead *wkhd;
 {
 	struct inoref *inoref;
 	struct jsegdep *jsegdep;
 	int needsj;
 
 	KASSERT((jaddref->ja_state & COMPLETE) == 0,
 	    ("cancel_jaddref: Canceling complete jaddref"));
 	if (jaddref->ja_state & (INPROGRESS | COMPLETE))
 		needsj = 1;
 	else
 		needsj = 0;
 	if (inodedep == NULL)
 		if (inodedep_lookup(jaddref->ja_list.wk_mp, jaddref->ja_ino,
 		    0, &inodedep) == 0)
 			panic("cancel_jaddref: Lost inodedep");
 	/*
 	 * We must adjust the nlink of any reference operation that follows
 	 * us so that it is consistent with the in-memory reference.  This
 	 * ensures that inode nlink rollbacks always have the correct link.
 	 */
 	if (needsj == 0) {
 		for (inoref = TAILQ_NEXT(&jaddref->ja_ref, if_deps); inoref;
 		    inoref = TAILQ_NEXT(inoref, if_deps)) {
 			if (inoref->if_state & GOINGAWAY)
 				break;
 			inoref->if_nlink--;
 		}
 	}
 	jsegdep = inoref_jseg(&jaddref->ja_ref);
 	if (jaddref->ja_state & NEWBLOCK)
 		move_newblock_dep(jaddref, inodedep);
 	wake_worklist(&jaddref->ja_list);
 	jaddref->ja_mkdir = NULL;
 	if (jaddref->ja_state & INPROGRESS) {
 		jaddref->ja_state &= ~INPROGRESS;
 		WORKLIST_REMOVE(&jaddref->ja_list);
 		jwork_insert(wkhd, jsegdep);
 	} else {
 		free_jsegdep(jsegdep);
 		if (jaddref->ja_state & DEPCOMPLETE)
 			remove_from_journal(&jaddref->ja_list);
 	}
 	jaddref->ja_state |= (GOINGAWAY | DEPCOMPLETE);
 	/*
 	 * Leave NEWBLOCK jaddrefs on the inodedep so handle_workitem_remove
 	 * can arrange for them to be freed with the bitmap.  Otherwise we
 	 * no longer need this addref attached to the inoreflst and it
 	 * will incorrectly adjust nlink if we leave it.
 	 */
 	if ((jaddref->ja_state & NEWBLOCK) == 0) {
 		TAILQ_REMOVE(&inodedep->id_inoreflst, &jaddref->ja_ref,
 		    if_deps);
 		jaddref->ja_state |= COMPLETE;
 		free_jaddref(jaddref);
 		return (needsj);
 	}
 	/*
 	 * Leave the head of the list for jsegdeps for fast merging.
 	 */
 	if (LIST_FIRST(wkhd) != NULL) {
 		jaddref->ja_state |= ONWORKLIST;
 		LIST_INSERT_AFTER(LIST_FIRST(wkhd), &jaddref->ja_list, wk_list);
 	} else
 		WORKLIST_INSERT(wkhd, &jaddref->ja_list);
 
 	return (needsj);
 }
 
 /* 
  * Attempt to free a jaddref structure when some work completes.  This
  * should only succeed once the entry is written and all dependencies have
  * been notified.
  */
 static void
 free_jaddref(jaddref)
 	struct jaddref *jaddref;
 {
 
 	if ((jaddref->ja_state & ALLCOMPLETE) != ALLCOMPLETE)
 		return;
 	if (jaddref->ja_ref.if_jsegdep)
 		panic("free_jaddref: segdep attached to jaddref %p(0x%X)\n",
 		    jaddref, jaddref->ja_state);
 	if (jaddref->ja_state & NEWBLOCK)
 		LIST_REMOVE(jaddref, ja_bmdeps);
 	if (jaddref->ja_state & (INPROGRESS | ONWORKLIST))
 		panic("free_jaddref: Bad state %p(0x%X)",
 		    jaddref, jaddref->ja_state);
 	if (jaddref->ja_mkdir != NULL)
 		panic("free_jaddref: Work pending, 0x%X\n", jaddref->ja_state);
 	WORKITEM_FREE(jaddref, D_JADDREF);
 }
 
 /*
  * Free a jremref structure once it has been written or discarded.
  */
 static void
 free_jremref(jremref)
 	struct jremref *jremref;
 {
 
 	if (jremref->jr_ref.if_jsegdep)
 		free_jsegdep(jremref->jr_ref.if_jsegdep);
 	if (jremref->jr_state & INPROGRESS)
 		panic("free_jremref: IO still pending");
 	WORKITEM_FREE(jremref, D_JREMREF);
 }
 
 /*
  * Free a jnewblk structure.
  */
 static void
 free_jnewblk(jnewblk)
 	struct jnewblk *jnewblk;
 {
 
 	if ((jnewblk->jn_state & ALLCOMPLETE) != ALLCOMPLETE)
 		return;
 	LIST_REMOVE(jnewblk, jn_deps);
 	if (jnewblk->jn_dep != NULL)
 		panic("free_jnewblk: Dependency still attached.");
 	WORKITEM_FREE(jnewblk, D_JNEWBLK);
 }
 
 /*
  * Cancel a jnewblk which has been been made redundant by frag extension.
  */
 static void
 cancel_jnewblk(jnewblk, wkhd)
 	struct jnewblk *jnewblk;
 	struct workhead *wkhd;
 {
 	struct jsegdep *jsegdep;
 
 	CTR1(KTR_SUJ, "cancel_jnewblk: blkno %jd", jnewblk->jn_blkno);
 	jsegdep = jnewblk->jn_jsegdep;
 	if (jnewblk->jn_jsegdep == NULL || jnewblk->jn_dep == NULL)
 		panic("cancel_jnewblk: Invalid state");
 	jnewblk->jn_jsegdep  = NULL;
 	jnewblk->jn_dep = NULL;
 	jnewblk->jn_state |= GOINGAWAY;
 	if (jnewblk->jn_state & INPROGRESS) {
 		jnewblk->jn_state &= ~INPROGRESS;
 		WORKLIST_REMOVE(&jnewblk->jn_list);
 		jwork_insert(wkhd, jsegdep);
 	} else {
 		free_jsegdep(jsegdep);
 		remove_from_journal(&jnewblk->jn_list);
 	}
 	wake_worklist(&jnewblk->jn_list);
 	WORKLIST_INSERT(wkhd, &jnewblk->jn_list);
 }
 
 static void
 free_jblkdep(jblkdep)
 	struct jblkdep *jblkdep;
 {
 
 	if (jblkdep->jb_list.wk_type == D_JFREEBLK)
 		WORKITEM_FREE(jblkdep, D_JFREEBLK);
 	else if (jblkdep->jb_list.wk_type == D_JTRUNC)
 		WORKITEM_FREE(jblkdep, D_JTRUNC);
 	else
 		panic("free_jblkdep: Unexpected type %s",
 		    TYPENAME(jblkdep->jb_list.wk_type));
 }
 
 /*
  * Free a single jseg once it is no longer referenced in memory or on
  * disk.  Reclaim journal blocks and dependencies waiting for the segment
  * to disappear.
  */
 static void
 free_jseg(jseg, jblocks)
 	struct jseg *jseg;
 	struct jblocks *jblocks;
 {
 	struct freework *freework;
 
 	/*
 	 * Free freework structures that were lingering to indicate freed
 	 * indirect blocks that forced journal write ordering on reallocate.
 	 */
 	while ((freework = LIST_FIRST(&jseg->js_indirs)) != NULL)
 		indirblk_remove(freework);
 	if (jblocks->jb_oldestseg == jseg)
 		jblocks->jb_oldestseg = TAILQ_NEXT(jseg, js_next);
 	TAILQ_REMOVE(&jblocks->jb_segs, jseg, js_next);
 	jblocks_free(jblocks, jseg->js_list.wk_mp, jseg->js_size);
 	KASSERT(LIST_EMPTY(&jseg->js_entries),
 	    ("free_jseg: Freed jseg has valid entries."));
 	WORKITEM_FREE(jseg, D_JSEG);
 }
 
 /*
  * Free all jsegs that meet the criteria for being reclaimed and update
  * oldestseg.
  */
 static void
 free_jsegs(jblocks)
 	struct jblocks *jblocks;
 {
 	struct jseg *jseg;
 
 	/*
 	 * Free only those jsegs which have none allocated before them to
 	 * preserve the journal space ordering.
 	 */
 	while ((jseg = TAILQ_FIRST(&jblocks->jb_segs)) != NULL) {
 		/*
 		 * Only reclaim space when nothing depends on this journal
 		 * set and another set has written that it is no longer
 		 * valid.
 		 */
 		if (jseg->js_refs != 0) {
 			jblocks->jb_oldestseg = jseg;
 			return;
 		}
 		if ((jseg->js_state & ALLCOMPLETE) != ALLCOMPLETE)
 			break;
 		if (jseg->js_seq > jblocks->jb_oldestwrseq)
 			break;
 		/*
 		 * We can free jsegs that didn't write entries when
 		 * oldestwrseq == js_seq.
 		 */
 		if (jseg->js_seq == jblocks->jb_oldestwrseq &&
 		    jseg->js_cnt != 0)
 			break;
 		free_jseg(jseg, jblocks);
 	}
 	/*
 	 * If we exited the loop above we still must discover the
 	 * oldest valid segment.
 	 */
 	if (jseg)
 		for (jseg = jblocks->jb_oldestseg; jseg != NULL;
 		     jseg = TAILQ_NEXT(jseg, js_next))
 			if (jseg->js_refs != 0)
 				break;
 	jblocks->jb_oldestseg = jseg;
 	/*
 	 * The journal has no valid records but some jsegs may still be
 	 * waiting on oldestwrseq to advance.  We force a small record
 	 * out to permit these lingering records to be reclaimed.
 	 */
 	if (jblocks->jb_oldestseg == NULL && !TAILQ_EMPTY(&jblocks->jb_segs))
 		jblocks->jb_needseg = 1;
 }
 
 /*
  * Release one reference to a jseg and free it if the count reaches 0.  This
  * should eventually reclaim journal space as well.
  */
 static void
 rele_jseg(jseg)
 	struct jseg *jseg;
 {
 
 	KASSERT(jseg->js_refs > 0,
 	    ("free_jseg: Invalid refcnt %d", jseg->js_refs));
 	if (--jseg->js_refs != 0)
 		return;
 	free_jsegs(jseg->js_jblocks);
 }
 
 /*
  * Release a jsegdep and decrement the jseg count.
  */
 static void
 free_jsegdep(jsegdep)
 	struct jsegdep *jsegdep;
 {
 
 	if (jsegdep->jd_seg)
 		rele_jseg(jsegdep->jd_seg);
 	WORKITEM_FREE(jsegdep, D_JSEGDEP);
 }
 
 /*
  * Wait for a journal item to make it to disk.  Initiate journal processing
  * if required.
  */
 static int
 jwait(wk, waitfor)
 	struct worklist *wk;
 	int waitfor;
 {
 
 	LOCK_OWNED(VFSTOUFS(wk->wk_mp));
 	/*
 	 * Blocking journal waits cause slow synchronous behavior.  Record
 	 * stats on the frequency of these blocking operations.
 	 */
 	if (waitfor == MNT_WAIT) {
 		stat_journal_wait++;
 		switch (wk->wk_type) {
 		case D_JREMREF:
 		case D_JMVREF:
 			stat_jwait_filepage++;
 			break;
 		case D_JTRUNC:
 		case D_JFREEBLK:
 			stat_jwait_freeblks++;
 			break;
 		case D_JNEWBLK:
 			stat_jwait_newblk++;
 			break;
 		case D_JADDREF:
 			stat_jwait_inode++;
 			break;
 		default:
 			break;
 		}
 	}
 	/*
 	 * If IO has not started we process the journal.  We can't mark the
 	 * worklist item as IOWAITING because we drop the lock while
 	 * processing the journal and the worklist entry may be freed after
 	 * this point.  The caller may call back in and re-issue the request.
 	 */
 	if ((wk->wk_state & INPROGRESS) == 0) {
 		softdep_process_journal(wk->wk_mp, wk, waitfor);
 		if (waitfor != MNT_WAIT)
 			return (EBUSY);
 		return (0);
 	}
 	if (waitfor != MNT_WAIT)
 		return (EBUSY);
 	wait_worklist(wk, "jwait");
 	return (0);
 }
 
 /*
  * Lookup an inodedep based on an inode pointer and set the nlinkdelta as
  * appropriate.  This is a convenience function to reduce duplicate code
  * for the setup and revert functions below.
  */
 static struct inodedep *
 inodedep_lookup_ip(ip)
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	int dflags;
 
 	KASSERT(ip->i_nlink >= ip->i_effnlink,
 	    ("inodedep_lookup_ip: bad delta"));
 	dflags = DEPALLOC;
 	if (IS_SNAPSHOT(ip))
 		dflags |= NODELAY;
 	(void) inodedep_lookup(UFSTOVFS(ip->i_ump), ip->i_number, dflags,
 	    &inodedep);
 	inodedep->id_nlinkdelta = ip->i_nlink - ip->i_effnlink;
 	KASSERT((inodedep->id_state & UNLINKED) == 0, ("inode unlinked"));
 
 	return (inodedep);
 }
 
 /*
  * Called prior to creating a new inode and linking it to a directory.  The
  * jaddref structure must already be allocated by softdep_setup_inomapdep
  * and it is discovered here so we can initialize the mode and update
  * nlinkdelta.
  */
 void
 softdep_setup_create(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_setup_create called on non-softdep filesystem"));
 	KASSERT(ip->i_nlink == 1,
 	    ("softdep_setup_create: Invalid link count."));
 	dvp = ITOV(dp);
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(ip);
 	if (DOINGSUJ(dvp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref != NULL && jaddref->ja_parent == dp->i_number,
 		    ("softdep_setup_create: No addref structure present."));
 	}
 	softdep_prelink(dvp, NULL);
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Create a jaddref structure to track the addition of a DOTDOT link when
  * we are reparenting an inode as part of a rename.  This jaddref will be
  * found by softdep_setup_directory_change.  Adjusts nlinkdelta for
  * non-journaling softdep.
  */
 void
 softdep_setup_dotdot_link(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct vnode *dvp;
 	struct vnode *vp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_setup_dotdot_link called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	vp = ITOV(ip);
 	jaddref = NULL;
 	/*
 	 * We don't set MKDIR_PARENT as this is not tied to a mkdir and
 	 * is used as a normal link would be.
 	 */
 	if (DOINGSUJ(dvp))
 		jaddref = newjaddref(ip, dp->i_number, DOTDOT_OFFSET,
 		    dp->i_effnlink - 1, dp->i_mode);
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(dp);
 	if (jaddref)
 		TAILQ_INSERT_TAIL(&inodedep->id_inoreflst, &jaddref->ja_ref,
 		    if_deps);
 	softdep_prelink(dvp, ITOV(ip));
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Create a jaddref structure to track a new link to an inode.  The directory
  * offset is not known until softdep_setup_directory_add or
  * softdep_setup_directory_change.  Adjusts nlinkdelta for non-journaling
  * softdep.
  */
 void
 softdep_setup_link(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_setup_link called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	jaddref = NULL;
 	if (DOINGSUJ(dvp))
 		jaddref = newjaddref(dp, ip->i_number, 0, ip->i_effnlink - 1,
 		    ip->i_mode);
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(ip);
 	if (jaddref)
 		TAILQ_INSERT_TAIL(&inodedep->id_inoreflst, &jaddref->ja_ref,
 		    if_deps);
 	softdep_prelink(dvp, ITOV(ip));
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called to create the jaddref structures to track . and .. references as
  * well as lookup and further initialize the incomplete jaddref created
  * by softdep_setup_inomapdep when the inode was allocated.  Adjusts
  * nlinkdelta for non-journaling softdep.
  */
 void
 softdep_setup_mkdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *dotdotaddref;
 	struct jaddref *dotaddref;
 	struct jaddref *jaddref;
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_setup_mkdir called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	dotaddref = dotdotaddref = NULL;
 	if (DOINGSUJ(dvp)) {
 		dotaddref = newjaddref(ip, ip->i_number, DOT_OFFSET, 1,
 		    ip->i_mode);
 		dotaddref->ja_state |= MKDIR_BODY;
 		dotdotaddref = newjaddref(ip, dp->i_number, DOTDOT_OFFSET,
 		    dp->i_effnlink - 1, dp->i_mode);
 		dotdotaddref->ja_state |= MKDIR_PARENT;
 	}
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(ip);
 	if (DOINGSUJ(dvp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref != NULL,
 		    ("softdep_setup_mkdir: No addref structure present."));
 		KASSERT(jaddref->ja_parent == dp->i_number, 
 		    ("softdep_setup_mkdir: bad parent %ju",
 		    (uintmax_t)jaddref->ja_parent));
 		TAILQ_INSERT_BEFORE(&jaddref->ja_ref, &dotaddref->ja_ref,
 		    if_deps);
 	}
 	inodedep = inodedep_lookup_ip(dp);
 	if (DOINGSUJ(dvp))
 		TAILQ_INSERT_TAIL(&inodedep->id_inoreflst,
 		    &dotdotaddref->ja_ref, if_deps);
 	softdep_prelink(ITOV(dp), NULL);
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called to track nlinkdelta of the inode and parent directories prior to
  * unlinking a directory.
  */
 void
 softdep_setup_rmdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_setup_rmdir called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	ACQUIRE_LOCK(dp->i_ump);
 	(void) inodedep_lookup_ip(ip);
 	(void) inodedep_lookup_ip(dp);
 	softdep_prelink(dvp, ITOV(ip));
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called to track nlinkdelta of the inode and parent directories prior to
  * unlink.
  */
 void
 softdep_setup_unlink(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_setup_unlink called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	ACQUIRE_LOCK(dp->i_ump);
 	(void) inodedep_lookup_ip(ip);
 	(void) inodedep_lookup_ip(dp);
 	softdep_prelink(dvp, ITOV(ip));
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called to release the journal structures created by a failed non-directory
  * creation.  Adjusts nlinkdelta for non-journaling softdep.
  */
 void
 softdep_revert_create(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_revert_create called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(ip);
 	if (DOINGSUJ(dvp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref->ja_parent == dp->i_number,
 		    ("softdep_revert_create: addref parent mismatch"));
 		cancel_jaddref(jaddref, inodedep, &inodedep->id_inowait);
 	}
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called to release the journal structures created by a failed link
  * addition.  Adjusts nlinkdelta for non-journaling softdep.
  */
 void
 softdep_revert_link(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_revert_link called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(ip);
 	if (DOINGSUJ(dvp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref->ja_parent == dp->i_number,
 		    ("softdep_revert_link: addref parent mismatch"));
 		cancel_jaddref(jaddref, inodedep, &inodedep->id_inowait);
 	}
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called to release the journal structures created by a failed mkdir
  * attempt.  Adjusts nlinkdelta for non-journaling softdep.
  */
 void
 softdep_revert_mkdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct jaddref *dotaddref;
 	struct vnode *dvp;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_revert_mkdir called on non-softdep filesystem"));
 	dvp = ITOV(dp);
 
 	ACQUIRE_LOCK(dp->i_ump);
 	inodedep = inodedep_lookup_ip(dp);
 	if (DOINGSUJ(dvp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref->ja_parent == ip->i_number,
 		    ("softdep_revert_mkdir: dotdot addref parent mismatch"));
 		cancel_jaddref(jaddref, inodedep, &inodedep->id_inowait);
 	}
 	inodedep = inodedep_lookup_ip(ip);
 	if (DOINGSUJ(dvp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref->ja_parent == dp->i_number,
 		    ("softdep_revert_mkdir: addref parent mismatch"));
 		dotaddref = (struct jaddref *)TAILQ_PREV(&jaddref->ja_ref,
 		    inoreflst, if_deps);
 		cancel_jaddref(jaddref, inodedep, &inodedep->id_inowait);
 		KASSERT(dotaddref->ja_parent == ip->i_number,
 		    ("softdep_revert_mkdir: dot addref parent mismatch"));
 		cancel_jaddref(dotaddref, inodedep, &inodedep->id_inowait);
 	}
 	FREE_LOCK(dp->i_ump);
 }
 
 /* 
  * Called to correct nlinkdelta after a failed rmdir.
  */
 void
 softdep_revert_rmdir(dp, ip)
 	struct inode *dp;
 	struct inode *ip;
 {
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(dp->i_ump)) != 0,
 	    ("softdep_revert_rmdir called on non-softdep filesystem"));
 	ACQUIRE_LOCK(dp->i_ump);
 	(void) inodedep_lookup_ip(ip);
 	(void) inodedep_lookup_ip(dp);
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Protecting the freemaps (or bitmaps).
  * 
  * To eliminate the need to execute fsck before mounting a filesystem
  * after a power failure, one must (conservatively) guarantee that the
  * on-disk copy of the bitmaps never indicate that a live inode or block is
  * free.  So, when a block or inode is allocated, the bitmap should be
  * updated (on disk) before any new pointers.  When a block or inode is
  * freed, the bitmap should not be updated until all pointers have been
  * reset.  The latter dependency is handled by the delayed de-allocation
  * approach described below for block and inode de-allocation.  The former
  * dependency is handled by calling the following procedure when a block or
  * inode is allocated. When an inode is allocated an "inodedep" is created
  * with its DEPCOMPLETE flag cleared until its bitmap is written to disk.
  * Each "inodedep" is also inserted into the hash indexing structure so
  * that any additional link additions can be made dependent on the inode
  * allocation.
  * 
  * The ufs filesystem maintains a number of free block counts (e.g., per
  * cylinder group, per cylinder and per <cylinder, rotational position> pair)
  * in addition to the bitmaps.  These counts are used to improve efficiency
  * during allocation and therefore must be consistent with the bitmaps.
  * There is no convenient way to guarantee post-crash consistency of these
  * counts with simple update ordering, for two main reasons: (1) The counts
  * and bitmaps for a single cylinder group block are not in the same disk
  * sector.  If a disk write is interrupted (e.g., by power failure), one may
  * be written and the other not.  (2) Some of the counts are located in the
  * superblock rather than the cylinder group block. So, we focus our soft
  * updates implementation on protecting the bitmaps. When mounting a
  * filesystem, we recompute the auxiliary counts from the bitmaps.
  */
 
 /*
  * Called just after updating the cylinder group block to allocate an inode.
  */
 void
 softdep_setup_inomapdep(bp, ip, newinum, mode)
 	struct buf *bp;		/* buffer for cylgroup block with inode map */
 	struct inode *ip;	/* inode related to allocation */
 	ino_t newinum;		/* new inode number being allocated */
 	int mode;
 {
 	struct inodedep *inodedep;
 	struct bmsafemap *bmsafemap;
 	struct jaddref *jaddref;
 	struct mount *mp;
 	struct fs *fs;
 
 	mp = UFSTOVFS(ip->i_ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_inomapdep called on non-softdep filesystem"));
 	fs = ip->i_ump->um_fs;
 	jaddref = NULL;
 
 	/*
 	 * Allocate the journal reference add structure so that the bitmap
 	 * can be dependent on it.
 	 */
 	if (MOUNTEDSUJ(mp)) {
 		jaddref = newjaddref(ip, newinum, 0, 0, mode);
 		jaddref->ja_state |= NEWBLOCK;
 	}
 
 	/*
 	 * Create a dependency for the newly allocated inode.
 	 * Panic if it already exists as something is seriously wrong.
 	 * Otherwise add it to the dependency list for the buffer holding
 	 * the cylinder group map from which it was allocated.
 	 *
 	 * We have to preallocate a bmsafemap entry in case it is needed
 	 * in bmsafemap_lookup since once we allocate the inodedep, we
 	 * have to finish initializing it before we can FREE_LOCK().
 	 * By preallocating, we avoid FREE_LOCK() while doing a malloc
 	 * in bmsafemap_lookup. We cannot call bmsafemap_lookup before
 	 * creating the inodedep as it can be freed during the time
 	 * that we FREE_LOCK() while allocating the inodedep. We must
 	 * call workitem_alloc() before entering the locked section as
 	 * it also acquires the lock and we must avoid trying doing so
 	 * recursively.
 	 */
 	bmsafemap = malloc(sizeof(struct bmsafemap),
 	    M_BMSAFEMAP, M_SOFTDEP_FLAGS);
 	workitem_alloc(&bmsafemap->sm_list, D_BMSAFEMAP, mp);
 	ACQUIRE_LOCK(ip->i_ump);
 	if ((inodedep_lookup(mp, newinum, DEPALLOC | NODELAY, &inodedep)))
 		panic("softdep_setup_inomapdep: dependency %p for new"
 		    "inode already exists", inodedep);
 	bmsafemap = bmsafemap_lookup(mp, bp, ino_to_cg(fs, newinum), bmsafemap);
 	if (jaddref) {
 		LIST_INSERT_HEAD(&bmsafemap->sm_jaddrefhd, jaddref, ja_bmdeps);
 		TAILQ_INSERT_TAIL(&inodedep->id_inoreflst, &jaddref->ja_ref,
 		    if_deps);
 	} else {
 		inodedep->id_state |= ONDEPLIST;
 		LIST_INSERT_HEAD(&bmsafemap->sm_inodedephd, inodedep, id_deps);
 	}
 	inodedep->id_bmsafemap = bmsafemap;
 	inodedep->id_state &= ~DEPCOMPLETE;
 	FREE_LOCK(ip->i_ump);
 }
 
 /*
  * Called just after updating the cylinder group block to
  * allocate block or fragment.
  */
 void
 softdep_setup_blkmapdep(bp, mp, newblkno, frags, oldfrags)
 	struct buf *bp;		/* buffer for cylgroup block with block map */
 	struct mount *mp;	/* filesystem doing allocation */
 	ufs2_daddr_t newblkno;	/* number of newly allocated block */
 	int frags;		/* Number of fragments. */
 	int oldfrags;		/* Previous number of fragments for extend. */
 {
 	struct newblk *newblk;
 	struct bmsafemap *bmsafemap;
 	struct jnewblk *jnewblk;
 	struct ufsmount *ump;
 	struct fs *fs;
 
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_blkmapdep called on non-softdep filesystem"));
 	ump = VFSTOUFS(mp);
 	fs = ump->um_fs;
 	jnewblk = NULL;
 	/*
 	 * Create a dependency for the newly allocated block.
 	 * Add it to the dependency list for the buffer holding
 	 * the cylinder group map from which it was allocated.
 	 */
 	if (MOUNTEDSUJ(mp)) {
 		jnewblk = malloc(sizeof(*jnewblk), M_JNEWBLK, M_SOFTDEP_FLAGS);
 		workitem_alloc(&jnewblk->jn_list, D_JNEWBLK, mp);
 		jnewblk->jn_jsegdep = newjsegdep(&jnewblk->jn_list);
 		jnewblk->jn_state = ATTACHED;
 		jnewblk->jn_blkno = newblkno;
 		jnewblk->jn_frags = frags;
 		jnewblk->jn_oldfrags = oldfrags;
 #ifdef SUJ_DEBUG
 		{
 			struct cg *cgp;
 			uint8_t *blksfree;
 			long bno;
 			int i;
 	
 			cgp = (struct cg *)bp->b_data;
 			blksfree = cg_blksfree(cgp);
 			bno = dtogd(fs, jnewblk->jn_blkno);
 			for (i = jnewblk->jn_oldfrags; i < jnewblk->jn_frags;
 			    i++) {
 				if (isset(blksfree, bno + i))
 					panic("softdep_setup_blkmapdep: "
 					    "free fragment %d from %d-%d "
 					    "state 0x%X dep %p", i,
 					    jnewblk->jn_oldfrags,
 					    jnewblk->jn_frags,
 					    jnewblk->jn_state,
 					    jnewblk->jn_dep);
 			}
 		}
 #endif
 	}
 
 	CTR3(KTR_SUJ,
 	    "softdep_setup_blkmapdep: blkno %jd frags %d oldfrags %d",
 	    newblkno, frags, oldfrags);
 	ACQUIRE_LOCK(ump);
 	if (newblk_lookup(mp, newblkno, DEPALLOC, &newblk) != 0)
 		panic("softdep_setup_blkmapdep: found block");
 	newblk->nb_bmsafemap = bmsafemap = bmsafemap_lookup(mp, bp,
 	    dtog(fs, newblkno), NULL);
 	if (jnewblk) {
 		jnewblk->jn_dep = (struct worklist *)newblk;
 		LIST_INSERT_HEAD(&bmsafemap->sm_jnewblkhd, jnewblk, jn_deps);
 	} else {
 		newblk->nb_state |= ONDEPLIST;
 		LIST_INSERT_HEAD(&bmsafemap->sm_newblkhd, newblk, nb_deps);
 	}
 	newblk->nb_bmsafemap = bmsafemap;
 	newblk->nb_jnewblk = jnewblk;
 	FREE_LOCK(ump);
 }
 
 #define	BMSAFEMAP_HASH(ump, cg) \
       (&(ump)->bmsafemap_hashtbl[(cg) & (ump)->bmsafemap_hash_size])
 
 static int
 bmsafemap_find(bmsafemaphd, cg, bmsafemapp)
 	struct bmsafemap_hashhead *bmsafemaphd;
 	int cg;
 	struct bmsafemap **bmsafemapp;
 {
 	struct bmsafemap *bmsafemap;
 
 	LIST_FOREACH(bmsafemap, bmsafemaphd, sm_hash)
 		if (bmsafemap->sm_cg == cg)
 			break;
 	if (bmsafemap) {
 		*bmsafemapp = bmsafemap;
 		return (1);
 	}
 	*bmsafemapp = NULL;
 
 	return (0);
 }
 
 /*
  * Find the bmsafemap associated with a cylinder group buffer.
  * If none exists, create one. The buffer must be locked when
  * this routine is called and this routine must be called with
  * the softdep lock held. To avoid giving up the lock while
  * allocating a new bmsafemap, a preallocated bmsafemap may be
  * provided. If it is provided but not needed, it is freed.
  */
 static struct bmsafemap *
 bmsafemap_lookup(mp, bp, cg, newbmsafemap)
 	struct mount *mp;
 	struct buf *bp;
 	int cg;
 	struct bmsafemap *newbmsafemap;
 {
 	struct bmsafemap_hashhead *bmsafemaphd;
 	struct bmsafemap *bmsafemap, *collision;
 	struct worklist *wk;
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	KASSERT(bp != NULL, ("bmsafemap_lookup: missing buffer"));
 	LIST_FOREACH(wk, &bp->b_dep, wk_list) {
 		if (wk->wk_type == D_BMSAFEMAP) {
 			if (newbmsafemap)
 				WORKITEM_FREE(newbmsafemap, D_BMSAFEMAP);
 			return (WK_BMSAFEMAP(wk));
 		}
 	}
 	bmsafemaphd = BMSAFEMAP_HASH(ump, cg);
 	if (bmsafemap_find(bmsafemaphd, cg, &bmsafemap) == 1) {
 		if (newbmsafemap)
 			WORKITEM_FREE(newbmsafemap, D_BMSAFEMAP);
 		return (bmsafemap);
 	}
 	if (newbmsafemap) {
 		bmsafemap = newbmsafemap;
 	} else {
 		FREE_LOCK(ump);
 		bmsafemap = malloc(sizeof(struct bmsafemap),
 			M_BMSAFEMAP, M_SOFTDEP_FLAGS);
 		workitem_alloc(&bmsafemap->sm_list, D_BMSAFEMAP, mp);
 		ACQUIRE_LOCK(ump);
 	}
 	bmsafemap->sm_buf = bp;
 	LIST_INIT(&bmsafemap->sm_inodedephd);
 	LIST_INIT(&bmsafemap->sm_inodedepwr);
 	LIST_INIT(&bmsafemap->sm_newblkhd);
 	LIST_INIT(&bmsafemap->sm_newblkwr);
 	LIST_INIT(&bmsafemap->sm_jaddrefhd);
 	LIST_INIT(&bmsafemap->sm_jnewblkhd);
 	LIST_INIT(&bmsafemap->sm_freehd);
 	LIST_INIT(&bmsafemap->sm_freewr);
 	if (bmsafemap_find(bmsafemaphd, cg, &collision) == 1) {
 		WORKITEM_FREE(bmsafemap, D_BMSAFEMAP);
 		return (collision);
 	}
 	bmsafemap->sm_cg = cg;
 	LIST_INSERT_HEAD(bmsafemaphd, bmsafemap, sm_hash);
 	LIST_INSERT_HEAD(&ump->softdep_dirtycg, bmsafemap, sm_next);
 	WORKLIST_INSERT(&bp->b_dep, &bmsafemap->sm_list);
 	return (bmsafemap);
 }
 
 /*
  * Direct block allocation dependencies.
  * 
  * When a new block is allocated, the corresponding disk locations must be
  * initialized (with zeros or new data) before the on-disk inode points to
  * them.  Also, the freemap from which the block was allocated must be
  * updated (on disk) before the inode's pointer. These two dependencies are
  * independent of each other and are needed for all file blocks and indirect
  * blocks that are pointed to directly by the inode.  Just before the
  * "in-core" version of the inode is updated with a newly allocated block
  * number, a procedure (below) is called to setup allocation dependency
  * structures.  These structures are removed when the corresponding
  * dependencies are satisfied or when the block allocation becomes obsolete
  * (i.e., the file is deleted, the block is de-allocated, or the block is a
  * fragment that gets upgraded).  All of these cases are handled in
  * procedures described later.
  * 
  * When a file extension causes a fragment to be upgraded, either to a larger
  * fragment or to a full block, the on-disk location may change (if the
  * previous fragment could not simply be extended). In this case, the old
  * fragment must be de-allocated, but not until after the inode's pointer has
  * been updated. In most cases, this is handled by later procedures, which
  * will construct a "freefrag" structure to be added to the workitem queue
  * when the inode update is complete (or obsolete).  The main exception to
  * this is when an allocation occurs while a pending allocation dependency
  * (for the same block pointer) remains.  This case is handled in the main
  * allocation dependency setup procedure by immediately freeing the
  * unreferenced fragments.
  */ 
 void 
 softdep_setup_allocdirect(ip, off, newblkno, oldblkno, newsize, oldsize, bp)
 	struct inode *ip;	/* inode to which block is being added */
 	ufs_lbn_t off;		/* block pointer within inode */
 	ufs2_daddr_t newblkno;	/* disk block number being added */
 	ufs2_daddr_t oldblkno;	/* previous block number, 0 unless frag */
 	long newsize;		/* size of new block */
 	long oldsize;		/* size of new block */
 	struct buf *bp;		/* bp for allocated block */
 {
 	struct allocdirect *adp, *oldadp;
 	struct allocdirectlst *adphead;
 	struct freefrag *freefrag;
 	struct inodedep *inodedep;
 	struct pagedep *pagedep;
 	struct jnewblk *jnewblk;
 	struct newblk *newblk;
 	struct mount *mp;
 	ufs_lbn_t lbn;
 
 	lbn = bp->b_lblkno;
 	mp = UFSTOVFS(ip->i_ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_allocdirect called on non-softdep filesystem"));
 	if (oldblkno && oldblkno != newblkno)
 		freefrag = newfreefrag(ip, oldblkno, oldsize, lbn);
 	else
 		freefrag = NULL;
 
 	CTR6(KTR_SUJ,
 	    "softdep_setup_allocdirect: ino %d blkno %jd oldblkno %jd "
 	    "off %jd newsize %ld oldsize %d",
 	    ip->i_number, newblkno, oldblkno, off, newsize, oldsize);
 	ACQUIRE_LOCK(ip->i_ump);
 	if (off >= NDADDR) {
 		if (lbn > 0)
 			panic("softdep_setup_allocdirect: bad lbn %jd, off %jd",
 			    lbn, off);
 		/* allocating an indirect block */
 		if (oldblkno != 0)
 			panic("softdep_setup_allocdirect: non-zero indir");
 	} else {
 		if (off != lbn)
 			panic("softdep_setup_allocdirect: lbn %jd != off %jd",
 			    lbn, off);
 		/*
 		 * Allocating a direct block.
 		 *
 		 * If we are allocating a directory block, then we must
 		 * allocate an associated pagedep to track additions and
 		 * deletions.
 		 */
 		if ((ip->i_mode & IFMT) == IFDIR)
 			pagedep_lookup(mp, bp, ip->i_number, off, DEPALLOC,
 			    &pagedep);
 	}
 	if (newblk_lookup(mp, newblkno, 0, &newblk) == 0)
 		panic("softdep_setup_allocdirect: lost block");
 	KASSERT(newblk->nb_list.wk_type == D_NEWBLK,
 	    ("softdep_setup_allocdirect: newblk already initialized"));
 	/*
 	 * Convert the newblk to an allocdirect.
 	 */
 	WORKITEM_REASSIGN(newblk, D_ALLOCDIRECT);
 	adp = (struct allocdirect *)newblk;
 	newblk->nb_freefrag = freefrag;
 	adp->ad_offset = off;
 	adp->ad_oldblkno = oldblkno;
 	adp->ad_newsize = newsize;
 	adp->ad_oldsize = oldsize;
 
 	/*
 	 * Finish initializing the journal.
 	 */
 	if ((jnewblk = newblk->nb_jnewblk) != NULL) {
 		jnewblk->jn_ino = ip->i_number;
 		jnewblk->jn_lbn = lbn;
 		add_to_journal(&jnewblk->jn_list);
 	}
 	if (freefrag && freefrag->ff_jdep != NULL &&
 	    freefrag->ff_jdep->wk_type == D_JFREEFRAG)
 		add_to_journal(freefrag->ff_jdep);
 	inodedep_lookup(mp, ip->i_number, DEPALLOC | NODELAY, &inodedep);
 	adp->ad_inodedep = inodedep;
 
 	WORKLIST_INSERT(&bp->b_dep, &newblk->nb_list);
 	/*
 	 * The list of allocdirects must be kept in sorted and ascending
 	 * order so that the rollback routines can quickly determine the
 	 * first uncommitted block (the size of the file stored on disk
 	 * ends at the end of the lowest committed fragment, or if there
 	 * are no fragments, at the end of the highest committed block).
 	 * Since files generally grow, the typical case is that the new
 	 * block is to be added at the end of the list. We speed this
 	 * special case by checking against the last allocdirect in the
 	 * list before laboriously traversing the list looking for the
 	 * insertion point.
 	 */
 	adphead = &inodedep->id_newinoupdt;
 	oldadp = TAILQ_LAST(adphead, allocdirectlst);
 	if (oldadp == NULL || oldadp->ad_offset <= off) {
 		/* insert at end of list */
 		TAILQ_INSERT_TAIL(adphead, adp, ad_next);
 		if (oldadp != NULL && oldadp->ad_offset == off)
 			allocdirect_merge(adphead, adp, oldadp);
 		FREE_LOCK(ip->i_ump);
 		return;
 	}
 	TAILQ_FOREACH(oldadp, adphead, ad_next) {
 		if (oldadp->ad_offset >= off)
 			break;
 	}
 	if (oldadp == NULL)
 		panic("softdep_setup_allocdirect: lost entry");
 	/* insert in middle of list */
 	TAILQ_INSERT_BEFORE(oldadp, adp, ad_next);
 	if (oldadp->ad_offset == off)
 		allocdirect_merge(adphead, adp, oldadp);
 
 	FREE_LOCK(ip->i_ump);
 }
 
 /*
  * Merge a newer and older journal record to be stored either in a
  * newblock or freefrag.  This handles aggregating journal records for
  * fragment allocation into a second record as well as replacing a
  * journal free with an aborted journal allocation.  A segment for the
  * oldest record will be placed on wkhd if it has been written.  If not
  * the segment for the newer record will suffice.
  */
 static struct worklist *
 jnewblk_merge(new, old, wkhd)
 	struct worklist *new;
 	struct worklist *old;
 	struct workhead *wkhd;
 {
 	struct jnewblk *njnewblk;
 	struct jnewblk *jnewblk;
 
 	/* Handle NULLs to simplify callers. */
 	if (new == NULL)
 		return (old);
 	if (old == NULL)
 		return (new);
 	/* Replace a jfreefrag with a jnewblk. */
 	if (new->wk_type == D_JFREEFRAG) {
 		if (WK_JNEWBLK(old)->jn_blkno != WK_JFREEFRAG(new)->fr_blkno)
 			panic("jnewblk_merge: blkno mismatch: %p, %p",
 			    old, new);
 		cancel_jfreefrag(WK_JFREEFRAG(new));
 		return (old);
 	}
 	if (old->wk_type != D_JNEWBLK || new->wk_type != D_JNEWBLK)
 		panic("jnewblk_merge: Bad type: old %d new %d\n",
 		    old->wk_type, new->wk_type);
 	/*
 	 * Handle merging of two jnewblk records that describe
 	 * different sets of fragments in the same block.
 	 */
 	jnewblk = WK_JNEWBLK(old);
 	njnewblk = WK_JNEWBLK(new);
 	if (jnewblk->jn_blkno != njnewblk->jn_blkno)
 		panic("jnewblk_merge: Merging disparate blocks.");
 	/*
 	 * The record may be rolled back in the cg.
 	 */
 	if (jnewblk->jn_state & UNDONE) {
 		jnewblk->jn_state &= ~UNDONE;
 		njnewblk->jn_state |= UNDONE;
 		njnewblk->jn_state &= ~ATTACHED;
 	}
 	/*
 	 * We modify the newer addref and free the older so that if neither
 	 * has been written the most up-to-date copy will be on disk.  If
 	 * both have been written but rolled back we only temporarily need
 	 * one of them to fix the bits when the cg write completes.
 	 */
 	jnewblk->jn_state |= ATTACHED | COMPLETE;
 	njnewblk->jn_oldfrags = jnewblk->jn_oldfrags;
 	cancel_jnewblk(jnewblk, wkhd);
 	WORKLIST_REMOVE(&jnewblk->jn_list);
 	free_jnewblk(jnewblk);
 	return (new);
 }
 
 /*
  * Replace an old allocdirect dependency with a newer one.
  * This routine must be called with splbio interrupts blocked.
  */
 static void
 allocdirect_merge(adphead, newadp, oldadp)
 	struct allocdirectlst *adphead;	/* head of list holding allocdirects */
 	struct allocdirect *newadp;	/* allocdirect being added */
 	struct allocdirect *oldadp;	/* existing allocdirect being checked */
 {
 	struct worklist *wk;
 	struct freefrag *freefrag;
 
 	freefrag = NULL;
 	LOCK_OWNED(VFSTOUFS(newadp->ad_list.wk_mp));
 	if (newadp->ad_oldblkno != oldadp->ad_newblkno ||
 	    newadp->ad_oldsize != oldadp->ad_newsize ||
 	    newadp->ad_offset >= NDADDR)
 		panic("%s %jd != new %jd || old size %ld != new %ld",
 		    "allocdirect_merge: old blkno",
 		    (intmax_t)newadp->ad_oldblkno,
 		    (intmax_t)oldadp->ad_newblkno,
 		    newadp->ad_oldsize, oldadp->ad_newsize);
 	newadp->ad_oldblkno = oldadp->ad_oldblkno;
 	newadp->ad_oldsize = oldadp->ad_oldsize;
 	/*
 	 * If the old dependency had a fragment to free or had never
 	 * previously had a block allocated, then the new dependency
 	 * can immediately post its freefrag and adopt the old freefrag.
 	 * This action is done by swapping the freefrag dependencies.
 	 * The new dependency gains the old one's freefrag, and the
 	 * old one gets the new one and then immediately puts it on
 	 * the worklist when it is freed by free_newblk. It is
 	 * not possible to do this swap when the old dependency had a
 	 * non-zero size but no previous fragment to free. This condition
 	 * arises when the new block is an extension of the old block.
 	 * Here, the first part of the fragment allocated to the new
 	 * dependency is part of the block currently claimed on disk by
 	 * the old dependency, so cannot legitimately be freed until the
 	 * conditions for the new dependency are fulfilled.
 	 */
 	freefrag = newadp->ad_freefrag;
 	if (oldadp->ad_freefrag != NULL || oldadp->ad_oldblkno == 0) {
 		newadp->ad_freefrag = oldadp->ad_freefrag;
 		oldadp->ad_freefrag = freefrag;
 	}
 	/*
 	 * If we are tracking a new directory-block allocation,
 	 * move it from the old allocdirect to the new allocdirect.
 	 */
 	if ((wk = LIST_FIRST(&oldadp->ad_newdirblk)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		if (!LIST_EMPTY(&oldadp->ad_newdirblk))
 			panic("allocdirect_merge: extra newdirblk");
 		WORKLIST_INSERT(&newadp->ad_newdirblk, wk);
 	}
 	TAILQ_REMOVE(adphead, oldadp, ad_next);
 	/*
 	 * We need to move any journal dependencies over to the freefrag
 	 * that releases this block if it exists.  Otherwise we are
 	 * extending an existing block and we'll wait until that is
 	 * complete to release the journal space and extend the
 	 * new journal to cover this old space as well.
 	 */
 	if (freefrag == NULL) {
 		if (oldadp->ad_newblkno != newadp->ad_newblkno)
 			panic("allocdirect_merge: %jd != %jd",
 			    oldadp->ad_newblkno, newadp->ad_newblkno);
 		newadp->ad_block.nb_jnewblk = (struct jnewblk *)
 		    jnewblk_merge(&newadp->ad_block.nb_jnewblk->jn_list, 
 		    &oldadp->ad_block.nb_jnewblk->jn_list,
 		    &newadp->ad_block.nb_jwork);
 		oldadp->ad_block.nb_jnewblk = NULL;
 		cancel_newblk(&oldadp->ad_block, NULL,
 		    &newadp->ad_block.nb_jwork);
 	} else {
 		wk = (struct worklist *) cancel_newblk(&oldadp->ad_block,
 		    &freefrag->ff_list, &freefrag->ff_jwork);
 		freefrag->ff_jdep = jnewblk_merge(freefrag->ff_jdep, wk,
 		    &freefrag->ff_jwork);
 	}
 	free_newblk(&oldadp->ad_block);
 }
 
 /*
  * Allocate a jfreefrag structure to journal a single block free.
  */
 static struct jfreefrag *
 newjfreefrag(freefrag, ip, blkno, size, lbn)
 	struct freefrag *freefrag;
 	struct inode *ip;
 	ufs2_daddr_t blkno;
 	long size;
 	ufs_lbn_t lbn;
 {
 	struct jfreefrag *jfreefrag;
 	struct fs *fs;
 
 	fs = ip->i_fs;
 	jfreefrag = malloc(sizeof(struct jfreefrag), M_JFREEFRAG,
 	    M_SOFTDEP_FLAGS);
 	workitem_alloc(&jfreefrag->fr_list, D_JFREEFRAG, UFSTOVFS(ip->i_ump));
 	jfreefrag->fr_jsegdep = newjsegdep(&jfreefrag->fr_list);
 	jfreefrag->fr_state = ATTACHED | DEPCOMPLETE;
 	jfreefrag->fr_ino = ip->i_number;
 	jfreefrag->fr_lbn = lbn;
 	jfreefrag->fr_blkno = blkno;
 	jfreefrag->fr_frags = numfrags(fs, size);
 	jfreefrag->fr_freefrag = freefrag;
 
 	return (jfreefrag);
 }
 
 /*
  * Allocate a new freefrag structure.
  */
 static struct freefrag *
 newfreefrag(ip, blkno, size, lbn)
 	struct inode *ip;
 	ufs2_daddr_t blkno;
 	long size;
 	ufs_lbn_t lbn;
 {
 	struct freefrag *freefrag;
 	struct fs *fs;
 
 	CTR4(KTR_SUJ, "newfreefrag: ino %d blkno %jd size %ld lbn %jd",
 	    ip->i_number, blkno, size, lbn);
 	fs = ip->i_fs;
 	if (fragnum(fs, blkno) + numfrags(fs, size) > fs->fs_frag)
 		panic("newfreefrag: frag size");
 	freefrag = malloc(sizeof(struct freefrag),
 	    M_FREEFRAG, M_SOFTDEP_FLAGS);
 	workitem_alloc(&freefrag->ff_list, D_FREEFRAG, UFSTOVFS(ip->i_ump));
 	freefrag->ff_state = ATTACHED;
 	LIST_INIT(&freefrag->ff_jwork);
 	freefrag->ff_inum = ip->i_number;
 	freefrag->ff_vtype = ITOV(ip)->v_type;
 	freefrag->ff_blkno = blkno;
 	freefrag->ff_fragsize = size;
 
 	if (MOUNTEDSUJ(UFSTOVFS(ip->i_ump))) {
 		freefrag->ff_jdep = (struct worklist *)
 		    newjfreefrag(freefrag, ip, blkno, size, lbn);
 	} else {
 		freefrag->ff_state |= DEPCOMPLETE;
 		freefrag->ff_jdep = NULL;
 	}
 
 	return (freefrag);
 }
 
 /*
  * This workitem de-allocates fragments that were replaced during
  * file block allocation.
  */
 static void 
 handle_workitem_freefrag(freefrag)
 	struct freefrag *freefrag;
 {
 	struct ufsmount *ump = VFSTOUFS(freefrag->ff_list.wk_mp);
 	struct workhead wkhd;
 
 	CTR3(KTR_SUJ,
 	    "handle_workitem_freefrag: ino %d blkno %jd size %ld",
 	    freefrag->ff_inum, freefrag->ff_blkno, freefrag->ff_fragsize);
 	/*
 	 * It would be illegal to add new completion items to the
 	 * freefrag after it was schedule to be done so it must be
 	 * safe to modify the list head here.
 	 */
 	LIST_INIT(&wkhd);
 	ACQUIRE_LOCK(ump);
 	LIST_SWAP(&freefrag->ff_jwork, &wkhd, worklist, wk_list);
 	/*
 	 * If the journal has not been written we must cancel it here.
 	 */
 	if (freefrag->ff_jdep) {
 		if (freefrag->ff_jdep->wk_type != D_JNEWBLK)
 			panic("handle_workitem_freefrag: Unexpected type %d\n",
 			    freefrag->ff_jdep->wk_type);
 		cancel_jnewblk(WK_JNEWBLK(freefrag->ff_jdep), &wkhd);
 	}
 	FREE_LOCK(ump);
 	ffs_blkfree(ump, ump->um_fs, ump->um_devvp, freefrag->ff_blkno,
 	   freefrag->ff_fragsize, freefrag->ff_inum, freefrag->ff_vtype, &wkhd);
 	ACQUIRE_LOCK(ump);
 	WORKITEM_FREE(freefrag, D_FREEFRAG);
 	FREE_LOCK(ump);
 }
 
 /*
  * Set up a dependency structure for an external attributes data block.
  * This routine follows much of the structure of softdep_setup_allocdirect.
  * See the description of softdep_setup_allocdirect above for details.
  */
 void 
 softdep_setup_allocext(ip, off, newblkno, oldblkno, newsize, oldsize, bp)
 	struct inode *ip;
 	ufs_lbn_t off;
 	ufs2_daddr_t newblkno;
 	ufs2_daddr_t oldblkno;
 	long newsize;
 	long oldsize;
 	struct buf *bp;
 {
 	struct allocdirect *adp, *oldadp;
 	struct allocdirectlst *adphead;
 	struct freefrag *freefrag;
 	struct inodedep *inodedep;
 	struct jnewblk *jnewblk;
 	struct newblk *newblk;
 	struct mount *mp;
 	ufs_lbn_t lbn;
 
 	mp = UFSTOVFS(ip->i_ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_allocext called on non-softdep filesystem"));
 	KASSERT(off < NXADDR, ("softdep_setup_allocext: lbn %lld > NXADDR",
 		    (long long)off));
 
 	lbn = bp->b_lblkno;
 	if (oldblkno && oldblkno != newblkno)
 		freefrag = newfreefrag(ip, oldblkno, oldsize, lbn);
 	else
 		freefrag = NULL;
 
 	ACQUIRE_LOCK(ip->i_ump);
 	if (newblk_lookup(mp, newblkno, 0, &newblk) == 0)
 		panic("softdep_setup_allocext: lost block");
 	KASSERT(newblk->nb_list.wk_type == D_NEWBLK,
 	    ("softdep_setup_allocext: newblk already initialized"));
 	/*
 	 * Convert the newblk to an allocdirect.
 	 */
 	WORKITEM_REASSIGN(newblk, D_ALLOCDIRECT);
 	adp = (struct allocdirect *)newblk;
 	newblk->nb_freefrag = freefrag;
 	adp->ad_offset = off;
 	adp->ad_oldblkno = oldblkno;
 	adp->ad_newsize = newsize;
 	adp->ad_oldsize = oldsize;
 	adp->ad_state |=  EXTDATA;
 
 	/*
 	 * Finish initializing the journal.
 	 */
 	if ((jnewblk = newblk->nb_jnewblk) != NULL) {
 		jnewblk->jn_ino = ip->i_number;
 		jnewblk->jn_lbn = lbn;
 		add_to_journal(&jnewblk->jn_list);
 	}
 	if (freefrag && freefrag->ff_jdep != NULL &&
 	    freefrag->ff_jdep->wk_type == D_JFREEFRAG)
 		add_to_journal(freefrag->ff_jdep);
 	inodedep_lookup(mp, ip->i_number, DEPALLOC | NODELAY, &inodedep);
 	adp->ad_inodedep = inodedep;
 
 	WORKLIST_INSERT(&bp->b_dep, &newblk->nb_list);
 	/*
 	 * The list of allocdirects must be kept in sorted and ascending
 	 * order so that the rollback routines can quickly determine the
 	 * first uncommitted block (the size of the file stored on disk
 	 * ends at the end of the lowest committed fragment, or if there
 	 * are no fragments, at the end of the highest committed block).
 	 * Since files generally grow, the typical case is that the new
 	 * block is to be added at the end of the list. We speed this
 	 * special case by checking against the last allocdirect in the
 	 * list before laboriously traversing the list looking for the
 	 * insertion point.
 	 */
 	adphead = &inodedep->id_newextupdt;
 	oldadp = TAILQ_LAST(adphead, allocdirectlst);
 	if (oldadp == NULL || oldadp->ad_offset <= off) {
 		/* insert at end of list */
 		TAILQ_INSERT_TAIL(adphead, adp, ad_next);
 		if (oldadp != NULL && oldadp->ad_offset == off)
 			allocdirect_merge(adphead, adp, oldadp);
 		FREE_LOCK(ip->i_ump);
 		return;
 	}
 	TAILQ_FOREACH(oldadp, adphead, ad_next) {
 		if (oldadp->ad_offset >= off)
 			break;
 	}
 	if (oldadp == NULL)
 		panic("softdep_setup_allocext: lost entry");
 	/* insert in middle of list */
 	TAILQ_INSERT_BEFORE(oldadp, adp, ad_next);
 	if (oldadp->ad_offset == off)
 		allocdirect_merge(adphead, adp, oldadp);
 	FREE_LOCK(ip->i_ump);
 }
 
 /*
  * Indirect block allocation dependencies.
  * 
  * The same dependencies that exist for a direct block also exist when
  * a new block is allocated and pointed to by an entry in a block of
  * indirect pointers. The undo/redo states described above are also
  * used here. Because an indirect block contains many pointers that
  * may have dependencies, a second copy of the entire in-memory indirect
  * block is kept. The buffer cache copy is always completely up-to-date.
  * The second copy, which is used only as a source for disk writes,
  * contains only the safe pointers (i.e., those that have no remaining
  * update dependencies). The second copy is freed when all pointers
  * are safe. The cache is not allowed to replace indirect blocks with
  * pending update dependencies. If a buffer containing an indirect
  * block with dependencies is written, these routines will mark it
  * dirty again. It can only be successfully written once all the
  * dependencies are removed. The ffs_fsync routine in conjunction with
  * softdep_sync_metadata work together to get all the dependencies
  * removed so that a file can be successfully written to disk. Three
  * procedures are used when setting up indirect block pointer
  * dependencies. The division is necessary because of the organization
  * of the "balloc" routine and because of the distinction between file
  * pages and file metadata blocks.
  */
 
 /*
  * Allocate a new allocindir structure.
  */
 static struct allocindir *
 newallocindir(ip, ptrno, newblkno, oldblkno, lbn)
 	struct inode *ip;	/* inode for file being extended */
 	int ptrno;		/* offset of pointer in indirect block */
 	ufs2_daddr_t newblkno;	/* disk block number being added */
 	ufs2_daddr_t oldblkno;	/* previous block number, 0 if none */
 	ufs_lbn_t lbn;
 {
 	struct newblk *newblk;
 	struct allocindir *aip;
 	struct freefrag *freefrag;
 	struct jnewblk *jnewblk;
 
 	if (oldblkno)
 		freefrag = newfreefrag(ip, oldblkno, ip->i_fs->fs_bsize, lbn);
 	else
 		freefrag = NULL;
 	ACQUIRE_LOCK(ip->i_ump);
 	if (newblk_lookup(UFSTOVFS(ip->i_ump), newblkno, 0, &newblk) == 0)
 		panic("new_allocindir: lost block");
 	KASSERT(newblk->nb_list.wk_type == D_NEWBLK,
 	    ("newallocindir: newblk already initialized"));
 	WORKITEM_REASSIGN(newblk, D_ALLOCINDIR);
 	newblk->nb_freefrag = freefrag;
 	aip = (struct allocindir *)newblk;
 	aip->ai_offset = ptrno;
 	aip->ai_oldblkno = oldblkno;
 	aip->ai_lbn = lbn;
 	if ((jnewblk = newblk->nb_jnewblk) != NULL) {
 		jnewblk->jn_ino = ip->i_number;
 		jnewblk->jn_lbn = lbn;
 		add_to_journal(&jnewblk->jn_list);
 	}
 	if (freefrag && freefrag->ff_jdep != NULL &&
 	    freefrag->ff_jdep->wk_type == D_JFREEFRAG)
 		add_to_journal(freefrag->ff_jdep);
 	return (aip);
 }
 
 /*
  * Called just before setting an indirect block pointer
  * to a newly allocated file page.
  */
 void
 softdep_setup_allocindir_page(ip, lbn, bp, ptrno, newblkno, oldblkno, nbp)
 	struct inode *ip;	/* inode for file being extended */
 	ufs_lbn_t lbn;		/* allocated block number within file */
 	struct buf *bp;		/* buffer with indirect blk referencing page */
 	int ptrno;		/* offset of pointer in indirect block */
 	ufs2_daddr_t newblkno;	/* disk block number being added */
 	ufs2_daddr_t oldblkno;	/* previous block number, 0 if none */
 	struct buf *nbp;	/* buffer holding allocated page */
 {
 	struct inodedep *inodedep;
 	struct freefrag *freefrag;
 	struct allocindir *aip;
 	struct pagedep *pagedep;
 	struct mount *mp;
 	int dflags;
 
 	mp = UFSTOVFS(ip->i_ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_allocindir_page called on non-softdep filesystem"));
 	KASSERT(lbn == nbp->b_lblkno,
 	    ("softdep_setup_allocindir_page: lbn %jd != lblkno %jd",
 	    lbn, bp->b_lblkno));
 	CTR4(KTR_SUJ,
 	    "softdep_setup_allocindir_page: ino %d blkno %jd oldblkno %jd "
 	    "lbn %jd", ip->i_number, newblkno, oldblkno, lbn);
 	ASSERT_VOP_LOCKED(ITOV(ip), "softdep_setup_allocindir_page");
 	aip = newallocindir(ip, ptrno, newblkno, oldblkno, lbn);
 	dflags = DEPALLOC;
 	if (IS_SNAPSHOT(ip))
 		dflags |= NODELAY;
 	(void) inodedep_lookup(mp, ip->i_number, dflags, &inodedep);
 	/*
 	 * If we are allocating a directory page, then we must
 	 * allocate an associated pagedep to track additions and
 	 * deletions.
 	 */
 	if ((ip->i_mode & IFMT) == IFDIR)
 		pagedep_lookup(mp, nbp, ip->i_number, lbn, DEPALLOC, &pagedep);
 	WORKLIST_INSERT(&nbp->b_dep, &aip->ai_block.nb_list);
 	freefrag = setup_allocindir_phase2(bp, ip, inodedep, aip, lbn);
 	FREE_LOCK(ip->i_ump);
 	if (freefrag)
 		handle_workitem_freefrag(freefrag);
 }
 
 /*
  * Called just before setting an indirect block pointer to a
  * newly allocated indirect block.
  */
 void
 softdep_setup_allocindir_meta(nbp, ip, bp, ptrno, newblkno)
 	struct buf *nbp;	/* newly allocated indirect block */
 	struct inode *ip;	/* inode for file being extended */
 	struct buf *bp;		/* indirect block referencing allocated block */
 	int ptrno;		/* offset of pointer in indirect block */
 	ufs2_daddr_t newblkno;	/* disk block number being added */
 {
 	struct inodedep *inodedep;
 	struct allocindir *aip;
 	ufs_lbn_t lbn;
 	int dflags;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_setup_allocindir_meta called on non-softdep filesystem"));
 	CTR3(KTR_SUJ,
 	    "softdep_setup_allocindir_meta: ino %d blkno %jd ptrno %d",
 	    ip->i_number, newblkno, ptrno);
 	lbn = nbp->b_lblkno;
 	ASSERT_VOP_LOCKED(ITOV(ip), "softdep_setup_allocindir_meta");
 	aip = newallocindir(ip, ptrno, newblkno, 0, lbn);
 	dflags = DEPALLOC;
 	if (IS_SNAPSHOT(ip))
 		dflags |= NODELAY;
 	inodedep_lookup(UFSTOVFS(ip->i_ump), ip->i_number, dflags, &inodedep);
 	WORKLIST_INSERT(&nbp->b_dep, &aip->ai_block.nb_list);
 	if (setup_allocindir_phase2(bp, ip, inodedep, aip, lbn))
 		panic("softdep_setup_allocindir_meta: Block already existed");
 	FREE_LOCK(ip->i_ump);
 }
 
 static void
 indirdep_complete(indirdep)
 	struct indirdep *indirdep;
 {
 	struct allocindir *aip;
 
 	LIST_REMOVE(indirdep, ir_next);
 	indirdep->ir_state |= DEPCOMPLETE;
 
 	while ((aip = LIST_FIRST(&indirdep->ir_completehd)) != NULL) {
 		LIST_REMOVE(aip, ai_next);
 		free_newblk(&aip->ai_block);
 	}
 	/*
 	 * If this indirdep is not attached to a buf it was simply waiting
 	 * on completion to clear completehd.  free_indirdep() asserts
 	 * that nothing is dangling.
 	 */
 	if ((indirdep->ir_state & ONWORKLIST) == 0)
 		free_indirdep(indirdep);
 }
 
 static struct indirdep *
 indirdep_lookup(mp, ip, bp)
 	struct mount *mp;
 	struct inode *ip;
 	struct buf *bp;
 {
 	struct indirdep *indirdep, *newindirdep;
 	struct newblk *newblk;
 	struct ufsmount *ump;
 	struct worklist *wk;
 	struct fs *fs;
 	ufs2_daddr_t blkno;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	indirdep = NULL;
 	newindirdep = NULL;
 	fs = ip->i_fs;
 	for (;;) {
 		LIST_FOREACH(wk, &bp->b_dep, wk_list) {
 			if (wk->wk_type != D_INDIRDEP)
 				continue;
 			indirdep = WK_INDIRDEP(wk);
 			break;
 		}
 		/* Found on the buffer worklist, no new structure to free. */
 		if (indirdep != NULL && newindirdep == NULL)
 			return (indirdep);
 		if (indirdep != NULL && newindirdep != NULL)
 			panic("indirdep_lookup: simultaneous create");
 		/* None found on the buffer and a new structure is ready. */
 		if (indirdep == NULL && newindirdep != NULL)
 			break;
 		/* None found and no new structure available. */
 		FREE_LOCK(ump);
 		newindirdep = malloc(sizeof(struct indirdep),
 		    M_INDIRDEP, M_SOFTDEP_FLAGS);
 		workitem_alloc(&newindirdep->ir_list, D_INDIRDEP, mp);
 		newindirdep->ir_state = ATTACHED;
 		if (ip->i_ump->um_fstype == UFS1)
 			newindirdep->ir_state |= UFS1FMT;
 		TAILQ_INIT(&newindirdep->ir_trunc);
 		newindirdep->ir_saveddata = NULL;
 		LIST_INIT(&newindirdep->ir_deplisthd);
 		LIST_INIT(&newindirdep->ir_donehd);
 		LIST_INIT(&newindirdep->ir_writehd);
 		LIST_INIT(&newindirdep->ir_completehd);
 		if (bp->b_blkno == bp->b_lblkno) {
 			ufs_bmaparray(bp->b_vp, bp->b_lblkno, &blkno, bp,
 			    NULL, NULL);
 			bp->b_blkno = blkno;
 		}
 		newindirdep->ir_freeblks = NULL;
 		newindirdep->ir_savebp =
 		    getblk(ip->i_devvp, bp->b_blkno, bp->b_bcount, 0, 0, 0);
 		newindirdep->ir_bp = bp;
 		BUF_KERNPROC(newindirdep->ir_savebp);
 		bcopy(bp->b_data, newindirdep->ir_savebp->b_data, bp->b_bcount);
 		ACQUIRE_LOCK(ump);
 	}
 	indirdep = newindirdep;
 	WORKLIST_INSERT(&bp->b_dep, &indirdep->ir_list);
 	/*
 	 * If the block is not yet allocated we don't set DEPCOMPLETE so
 	 * that we don't free dependencies until the pointers are valid.
 	 * This could search b_dep for D_ALLOCDIRECT/D_ALLOCINDIR rather
 	 * than using the hash.
 	 */
 	if (newblk_lookup(mp, dbtofsb(fs, bp->b_blkno), 0, &newblk))
 		LIST_INSERT_HEAD(&newblk->nb_indirdeps, indirdep, ir_next);
 	else
 		indirdep->ir_state |= DEPCOMPLETE;
 	return (indirdep);
 }
 
 /*
  * Called to finish the allocation of the "aip" allocated
  * by one of the two routines above.
  */
 static struct freefrag *
 setup_allocindir_phase2(bp, ip, inodedep, aip, lbn)
 	struct buf *bp;		/* in-memory copy of the indirect block */
 	struct inode *ip;	/* inode for file being extended */
 	struct inodedep *inodedep; /* Inodedep for ip */
 	struct allocindir *aip;	/* allocindir allocated by the above routines */
 	ufs_lbn_t lbn;		/* Logical block number for this block. */
 {
 	struct fs *fs;
 	struct indirdep *indirdep;
 	struct allocindir *oldaip;
 	struct freefrag *freefrag;
 	struct mount *mp;
 
 	LOCK_OWNED(ip->i_ump);
 	mp = UFSTOVFS(ip->i_ump);
 	fs = ip->i_fs;
 	if (bp->b_lblkno >= 0)
 		panic("setup_allocindir_phase2: not indir blk");
 	KASSERT(aip->ai_offset >= 0 && aip->ai_offset < NINDIR(fs),
 	    ("setup_allocindir_phase2: Bad offset %d", aip->ai_offset));
 	indirdep = indirdep_lookup(mp, ip, bp);
 	KASSERT(indirdep->ir_savebp != NULL,
 	    ("setup_allocindir_phase2 NULL ir_savebp"));
 	aip->ai_indirdep = indirdep;
 	/*
 	 * Check for an unwritten dependency for this indirect offset.  If
 	 * there is, merge the old dependency into the new one.  This happens
 	 * as a result of reallocblk only.
 	 */
 	freefrag = NULL;
 	if (aip->ai_oldblkno != 0) {
 		LIST_FOREACH(oldaip, &indirdep->ir_deplisthd, ai_next) {
 			if (oldaip->ai_offset == aip->ai_offset) {
 				freefrag = allocindir_merge(aip, oldaip);
 				goto done;
 			}
 		}
 		LIST_FOREACH(oldaip, &indirdep->ir_donehd, ai_next) {
 			if (oldaip->ai_offset == aip->ai_offset) {
 				freefrag = allocindir_merge(aip, oldaip);
 				goto done;
 			}
 		}
 	}
 done:
 	LIST_INSERT_HEAD(&indirdep->ir_deplisthd, aip, ai_next);
 	return (freefrag);
 }
 
 /*
  * Merge two allocindirs which refer to the same block.  Move newblock
  * dependencies and setup the freefrags appropriately.
  */
 static struct freefrag *
 allocindir_merge(aip, oldaip)
 	struct allocindir *aip;
 	struct allocindir *oldaip;
 {
 	struct freefrag *freefrag;
 	struct worklist *wk;
 
 	if (oldaip->ai_newblkno != aip->ai_oldblkno)
 		panic("allocindir_merge: blkno");
 	aip->ai_oldblkno = oldaip->ai_oldblkno;
 	freefrag = aip->ai_freefrag;
 	aip->ai_freefrag = oldaip->ai_freefrag;
 	oldaip->ai_freefrag = NULL;
 	KASSERT(freefrag != NULL, ("setup_allocindir_phase2: No freefrag"));
 	/*
 	 * If we are tracking a new directory-block allocation,
 	 * move it from the old allocindir to the new allocindir.
 	 */
 	if ((wk = LIST_FIRST(&oldaip->ai_newdirblk)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		if (!LIST_EMPTY(&oldaip->ai_newdirblk))
 			panic("allocindir_merge: extra newdirblk");
 		WORKLIST_INSERT(&aip->ai_newdirblk, wk);
 	}
 	/*
 	 * We can skip journaling for this freefrag and just complete
 	 * any pending journal work for the allocindir that is being
 	 * removed after the freefrag completes.
 	 */
 	if (freefrag->ff_jdep)
 		cancel_jfreefrag(WK_JFREEFRAG(freefrag->ff_jdep));
 	LIST_REMOVE(oldaip, ai_next);
 	freefrag->ff_jdep = (struct worklist *)cancel_newblk(&oldaip->ai_block,
 	    &freefrag->ff_list, &freefrag->ff_jwork);
 	free_newblk(&oldaip->ai_block);
 
 	return (freefrag);
 }
 
 static inline void
 setup_freedirect(freeblks, ip, i, needj)
 	struct freeblks *freeblks;
 	struct inode *ip;
 	int i;
 	int needj;
 {
 	ufs2_daddr_t blkno;
 	int frags;
 
 	blkno = DIP(ip, i_db[i]);
 	if (blkno == 0)
 		return;
 	DIP_SET(ip, i_db[i], 0);
 	frags = sblksize(ip->i_fs, ip->i_size, i);
 	frags = numfrags(ip->i_fs, frags);
 	newfreework(ip->i_ump, freeblks, NULL, i, blkno, frags, 0, needj);
 }
 
 static inline void
 setup_freeext(freeblks, ip, i, needj)
 	struct freeblks *freeblks;
 	struct inode *ip;
 	int i;
 	int needj;
 {
 	ufs2_daddr_t blkno;
 	int frags;
 
 	blkno = ip->i_din2->di_extb[i];
 	if (blkno == 0)
 		return;
 	ip->i_din2->di_extb[i] = 0;
 	frags = sblksize(ip->i_fs, ip->i_din2->di_extsize, i);
 	frags = numfrags(ip->i_fs, frags);
 	newfreework(ip->i_ump, freeblks, NULL, -1 - i, blkno, frags, 0, needj);
 }
 
 static inline void
 setup_freeindir(freeblks, ip, i, lbn, needj)
 	struct freeblks *freeblks;
 	struct inode *ip;
 	int i;
 	ufs_lbn_t lbn;
 	int needj;
 {
 	ufs2_daddr_t blkno;
 
 	blkno = DIP(ip, i_ib[i]);
 	if (blkno == 0)
 		return;
 	DIP_SET(ip, i_ib[i], 0);
 	newfreework(ip->i_ump, freeblks, NULL, lbn, blkno, ip->i_fs->fs_frag,
 	    0, needj);
 }
 
 static inline struct freeblks *
 newfreeblks(mp, ip)
 	struct mount *mp;
 	struct inode *ip;
 {
 	struct freeblks *freeblks;
 
 	freeblks = malloc(sizeof(struct freeblks),
 		M_FREEBLKS, M_SOFTDEP_FLAGS|M_ZERO);
 	workitem_alloc(&freeblks->fb_list, D_FREEBLKS, mp);
 	LIST_INIT(&freeblks->fb_jblkdephd);
 	LIST_INIT(&freeblks->fb_jwork);
 	freeblks->fb_ref = 0;
 	freeblks->fb_cgwait = 0;
 	freeblks->fb_state = ATTACHED;
 	freeblks->fb_uid = ip->i_uid;
 	freeblks->fb_inum = ip->i_number;
 	freeblks->fb_vtype = ITOV(ip)->v_type;
 	freeblks->fb_modrev = DIP(ip, i_modrev);
 	freeblks->fb_devvp = ip->i_devvp;
 	freeblks->fb_chkcnt = 0;
 	freeblks->fb_len = 0;
 
 	return (freeblks);
 }
 
 static void
 trunc_indirdep(indirdep, freeblks, bp, off)
 	struct indirdep *indirdep;
 	struct freeblks *freeblks;
 	struct buf *bp;
 	int off;
 {
 	struct allocindir *aip, *aipn;
 
 	/*
 	 * The first set of allocindirs won't be in savedbp.
 	 */
 	LIST_FOREACH_SAFE(aip, &indirdep->ir_deplisthd, ai_next, aipn)
 		if (aip->ai_offset > off)
 			cancel_allocindir(aip, bp, freeblks, 1);
 	LIST_FOREACH_SAFE(aip, &indirdep->ir_donehd, ai_next, aipn)
 		if (aip->ai_offset > off)
 			cancel_allocindir(aip, bp, freeblks, 1);
 	/*
 	 * These will exist in savedbp.
 	 */
 	LIST_FOREACH_SAFE(aip, &indirdep->ir_writehd, ai_next, aipn)
 		if (aip->ai_offset > off)
 			cancel_allocindir(aip, NULL, freeblks, 0);
 	LIST_FOREACH_SAFE(aip, &indirdep->ir_completehd, ai_next, aipn)
 		if (aip->ai_offset > off)
 			cancel_allocindir(aip, NULL, freeblks, 0);
 }
 
 /*
  * Follow the chain of indirects down to lastlbn creating a freework
  * structure for each.  This will be used to start indir_trunc() at
  * the right offset and create the journal records for the parrtial
  * truncation.  A second step will handle the truncated dependencies.
  */
 static int
 setup_trunc_indir(freeblks, ip, lbn, lastlbn, blkno)
 	struct freeblks *freeblks;
 	struct inode *ip;
 	ufs_lbn_t lbn;
 	ufs_lbn_t lastlbn;
 	ufs2_daddr_t blkno;
 {
 	struct indirdep *indirdep;
 	struct indirdep *indirn;
 	struct freework *freework;
 	struct newblk *newblk;
 	struct mount *mp;
 	struct buf *bp;
 	uint8_t *start;
 	uint8_t *end;
 	ufs_lbn_t lbnadd;
 	int level;
 	int error;
 	int off;
 
 
 	freework = NULL;
 	if (blkno == 0)
 		return (0);
 	mp = freeblks->fb_list.wk_mp;
 	bp = getblk(ITOV(ip), lbn, mp->mnt_stat.f_iosize, 0, 0, 0);
 	if ((bp->b_flags & B_CACHE) == 0) {
 		bp->b_blkno = blkptrtodb(VFSTOUFS(mp), blkno);
 		bp->b_iocmd = BIO_READ;
 		bp->b_flags &= ~B_INVAL;
 		bp->b_ioflags &= ~BIO_ERROR;
 		vfs_busy_pages(bp, 0);
 		bp->b_iooffset = dbtob(bp->b_blkno);
 		bstrategy(bp);
 		curthread->td_ru.ru_inblock++;
 		error = bufwait(bp);
 		if (error) {
 			brelse(bp);
 			return (error);
 		}
 	}
 	level = lbn_level(lbn);
 	lbnadd = lbn_offset(ip->i_fs, level);
 	/*
 	 * Compute the offset of the last block we want to keep.  Store
 	 * in the freework the first block we want to completely free.
 	 */
 	off = (lastlbn - -(lbn + level)) / lbnadd;
 	if (off + 1 == NINDIR(ip->i_fs))
 		goto nowork;
 	freework = newfreework(ip->i_ump, freeblks, NULL, lbn, blkno, 0, off+1,
 	    0);
 	/*
 	 * Link the freework into the indirdep.  This will prevent any new
 	 * allocations from proceeding until we are finished with the
 	 * truncate and the block is written.
 	 */
 	ACQUIRE_LOCK(ip->i_ump);
 	indirdep = indirdep_lookup(mp, ip, bp);
 	if (indirdep->ir_freeblks)
 		panic("setup_trunc_indir: indirdep already truncated.");
 	TAILQ_INSERT_TAIL(&indirdep->ir_trunc, freework, fw_next);
 	freework->fw_indir = indirdep;
 	/*
 	 * Cancel any allocindirs that will not make it to disk.
 	 * We have to do this for all copies of the indirdep that
 	 * live on this newblk.
 	 */
 	if ((indirdep->ir_state & DEPCOMPLETE) == 0) {
 		newblk_lookup(mp, dbtofsb(ip->i_fs, bp->b_blkno), 0, &newblk);
 		LIST_FOREACH(indirn, &newblk->nb_indirdeps, ir_next)
 			trunc_indirdep(indirn, freeblks, bp, off);
 	} else
 		trunc_indirdep(indirdep, freeblks, bp, off);
 	FREE_LOCK(ip->i_ump);
 	/*
 	 * Creation is protected by the buf lock. The saveddata is only
 	 * needed if a full truncation follows a partial truncation but it
 	 * is difficult to allocate in that case so we fetch it anyway.
 	 */
 	if (indirdep->ir_saveddata == NULL)
 		indirdep->ir_saveddata = malloc(bp->b_bcount, M_INDIRDEP,
 		    M_SOFTDEP_FLAGS);
 nowork:
 	/* Fetch the blkno of the child and the zero start offset. */
 	if (ip->i_ump->um_fstype == UFS1) {
 		blkno = ((ufs1_daddr_t *)bp->b_data)[off];
 		start = (uint8_t *)&((ufs1_daddr_t *)bp->b_data)[off+1];
 	} else {
 		blkno = ((ufs2_daddr_t *)bp->b_data)[off];
 		start = (uint8_t *)&((ufs2_daddr_t *)bp->b_data)[off+1];
 	}
 	if (freework) {
 		/* Zero the truncated pointers. */
 		end = bp->b_data + bp->b_bcount;
 		bzero(start, end - start);
 		bdwrite(bp);
 	} else
 		bqrelse(bp);
 	if (level == 0)
 		return (0);
 	lbn++; /* adjust level */
 	lbn -= (off * lbnadd);
 	return setup_trunc_indir(freeblks, ip, lbn, lastlbn, blkno);
 }
 
 /*
  * Complete the partial truncation of an indirect block setup by
  * setup_trunc_indir().  This zeros the truncated pointers in the saved
  * copy and writes them to disk before the freeblks is allowed to complete.
  */
 static void
 complete_trunc_indir(freework)
 	struct freework *freework;
 {
 	struct freework *fwn;
 	struct indirdep *indirdep;
 	struct ufsmount *ump;
 	struct buf *bp;
 	uintptr_t start;
 	int count;
 
 	ump = VFSTOUFS(freework->fw_list.wk_mp);
 	LOCK_OWNED(ump);
 	indirdep = freework->fw_indir;
 	for (;;) {
 		bp = indirdep->ir_bp;
 		/* See if the block was discarded. */
 		if (bp == NULL)
 			break;
 		/* Inline part of getdirtybuf().  We dont want bremfree. */
 		if (BUF_LOCK(bp, LK_EXCLUSIVE | LK_NOWAIT, NULL) == 0)
 			break;
 		if (BUF_LOCK(bp, LK_EXCLUSIVE | LK_SLEEPFAIL | LK_INTERLOCK,
 		    LOCK_PTR(ump)) == 0)
 			BUF_UNLOCK(bp);
 		ACQUIRE_LOCK(ump);
 	}
 	freework->fw_state |= DEPCOMPLETE;
 	TAILQ_REMOVE(&indirdep->ir_trunc, freework, fw_next);
 	/*
 	 * Zero the pointers in the saved copy.
 	 */
 	if (indirdep->ir_state & UFS1FMT)
 		start = sizeof(ufs1_daddr_t);
 	else
 		start = sizeof(ufs2_daddr_t);
 	start *= freework->fw_start;
 	count = indirdep->ir_savebp->b_bcount - start;
 	start += (uintptr_t)indirdep->ir_savebp->b_data;
 	bzero((char *)start, count);
 	/*
 	 * We need to start the next truncation in the list if it has not
 	 * been started yet.
 	 */
 	fwn = TAILQ_FIRST(&indirdep->ir_trunc);
 	if (fwn != NULL) {
 		if (fwn->fw_freeblks == indirdep->ir_freeblks)
 			TAILQ_REMOVE(&indirdep->ir_trunc, fwn, fw_next);
 		if ((fwn->fw_state & ONWORKLIST) == 0)
 			freework_enqueue(fwn);
 	}
 	/*
 	 * If bp is NULL the block was fully truncated, restore
 	 * the saved block list otherwise free it if it is no
 	 * longer needed.
 	 */
 	if (TAILQ_EMPTY(&indirdep->ir_trunc)) {
 		if (bp == NULL)
 			bcopy(indirdep->ir_saveddata,
 			    indirdep->ir_savebp->b_data,
 			    indirdep->ir_savebp->b_bcount);
 		free(indirdep->ir_saveddata, M_INDIRDEP);
 		indirdep->ir_saveddata = NULL;
 	}
 	/*
 	 * When bp is NULL there is a full truncation pending.  We
 	 * must wait for this full truncation to be journaled before
 	 * we can release this freework because the disk pointers will
 	 * never be written as zero.
 	 */
 	if (bp == NULL)  {
 		if (LIST_EMPTY(&indirdep->ir_freeblks->fb_jblkdephd))
 			handle_written_freework(freework);
 		else
 			WORKLIST_INSERT(&indirdep->ir_freeblks->fb_freeworkhd,
 			   &freework->fw_list);
 	} else {
 		/* Complete when the real copy is written. */
 		WORKLIST_INSERT(&bp->b_dep, &freework->fw_list);
 		BUF_UNLOCK(bp);
 	}
 }
 
 /*
  * Calculate the number of blocks we are going to release where datablocks
  * is the current total and length is the new file size.
  */
 static ufs2_daddr_t
 blkcount(fs, datablocks, length)
 	struct fs *fs;
 	ufs2_daddr_t datablocks;
 	off_t length;
 {
 	off_t totblks, numblks;
 
 	totblks = 0;
 	numblks = howmany(length, fs->fs_bsize);
 	if (numblks <= NDADDR) {
 		totblks = howmany(length, fs->fs_fsize);
 		goto out;
 	}
         totblks = blkstofrags(fs, numblks);
 	numblks -= NDADDR;
 	/*
 	 * Count all single, then double, then triple indirects required.
 	 * Subtracting one indirects worth of blocks for each pass
 	 * acknowledges one of each pointed to by the inode.
 	 */
 	for (;;) {
 		totblks += blkstofrags(fs, howmany(numblks, NINDIR(fs)));
 		numblks -= NINDIR(fs);
 		if (numblks <= 0)
 			break;
 		numblks = howmany(numblks, NINDIR(fs));
 	}
 out:
 	totblks = fsbtodb(fs, totblks);
 	/*
 	 * Handle sparse files.  We can't reclaim more blocks than the inode
 	 * references.  We will correct it later in handle_complete_freeblks()
 	 * when we know the real count.
 	 */
 	if (totblks > datablocks)
 		return (0);
 	return (datablocks - totblks);
 }
 
 /*
  * Handle freeblocks for journaled softupdate filesystems.
  *
  * Contrary to normal softupdates, we must preserve the block pointers in
  * indirects until their subordinates are free.  This is to avoid journaling
  * every block that is freed which may consume more space than the journal
  * itself.  The recovery program will see the free block journals at the
  * base of the truncated area and traverse them to reclaim space.  The
  * pointers in the inode may be cleared immediately after the journal
  * records are written because each direct and indirect pointer in the
  * inode is recorded in a journal.  This permits full truncation to proceed
  * asynchronously.  The write order is journal -> inode -> cgs -> indirects.
  *
  * The algorithm is as follows:
  * 1) Traverse the in-memory state and create journal entries to release
  *    the relevant blocks and full indirect trees.
  * 2) Traverse the indirect block chain adding partial truncation freework
  *    records to indirects in the path to lastlbn.  The freework will
  *    prevent new allocation dependencies from being satisfied in this
  *    indirect until the truncation completes.
  * 3) Read and lock the inode block, performing an update with the new size
  *    and pointers.  This prevents truncated data from becoming valid on
  *    disk through step 4.
  * 4) Reap unsatisfied dependencies that are beyond the truncated area,
  *    eliminate journal work for those records that do not require it.
  * 5) Schedule the journal records to be written followed by the inode block.
  * 6) Allocate any necessary frags for the end of file.
  * 7) Zero any partially truncated blocks.
  *
  * From this truncation proceeds asynchronously using the freework and
  * indir_trunc machinery.  The file will not be extended again into a
  * partially truncated indirect block until all work is completed but
  * the normal dependency mechanism ensures that it is rolled back/forward
  * as appropriate.  Further truncation may occur without delay and is
  * serialized in indir_trunc().
  */
 void
 softdep_journal_freeblocks(ip, cred, length, flags)
 	struct inode *ip;	/* The inode whose length is to be reduced */
 	struct ucred *cred;
 	off_t length;		/* The new length for the file */
 	int flags;		/* IO_EXT and/or IO_NORMAL */
 {
 	struct freeblks *freeblks, *fbn;
 	struct worklist *wk, *wkn;
 	struct inodedep *inodedep;
 	struct jblkdep *jblkdep;
 	struct allocdirect *adp, *adpn;
 	struct ufsmount *ump;
 	struct fs *fs;
 	struct buf *bp;
 	struct vnode *vp;
 	struct mount *mp;
 	ufs2_daddr_t extblocks, datablocks;
 	ufs_lbn_t tmpval, lbn, lastlbn;
 	int frags, lastoff, iboff, allocblock, needj, dflags, error, i;
 
 	fs = ip->i_fs;
 	ump = ip->i_ump;
 	mp = UFSTOVFS(ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_journal_freeblocks called on non-softdep filesystem"));
 	vp = ITOV(ip);
 	needj = 1;
 	iboff = -1;
 	allocblock = 0;
 	extblocks = 0;
 	datablocks = 0;
 	frags = 0;
 	freeblks = newfreeblks(mp, ip);
 	ACQUIRE_LOCK(ump);
 	/*
 	 * If we're truncating a removed file that will never be written
 	 * we don't need to journal the block frees.  The canceled journals
 	 * for the allocations will suffice.
 	 */
 	dflags = DEPALLOC;
 	if (IS_SNAPSHOT(ip))
 		dflags |= NODELAY;
 	inodedep_lookup(mp, ip->i_number, dflags, &inodedep);
 	if ((inodedep->id_state & (UNLINKED | DEPCOMPLETE)) == UNLINKED &&
 	    length == 0)
 		needj = 0;
 	CTR3(KTR_SUJ, "softdep_journal_freeblks: ip %d length %ld needj %d",
 	    ip->i_number, length, needj);
 	FREE_LOCK(ump);
 	/*
 	 * Calculate the lbn that we are truncating to.  This results in -1
 	 * if we're truncating the 0 bytes.  So it is the last lbn we want
 	 * to keep, not the first lbn we want to truncate.
 	 */
 	lastlbn = lblkno(fs, length + fs->fs_bsize - 1) - 1;
 	lastoff = blkoff(fs, length);
 	/*
 	 * Compute frags we are keeping in lastlbn.  0 means all.
 	 */
 	if (lastlbn >= 0 && lastlbn < NDADDR) {
 		frags = fragroundup(fs, lastoff);
 		/* adp offset of last valid allocdirect. */
 		iboff = lastlbn;
 	} else if (lastlbn > 0)
 		iboff = NDADDR;
 	if (fs->fs_magic == FS_UFS2_MAGIC)
 		extblocks = btodb(fragroundup(fs, ip->i_din2->di_extsize));
 	/*
 	 * Handle normal data blocks and indirects.  This section saves
 	 * values used after the inode update to complete frag and indirect
 	 * truncation.
 	 */
 	if ((flags & IO_NORMAL) != 0) {
 		/*
 		 * Handle truncation of whole direct and indirect blocks.
 		 */
 		for (i = iboff + 1; i < NDADDR; i++)
 			setup_freedirect(freeblks, ip, i, needj);
 		for (i = 0, tmpval = NINDIR(fs), lbn = NDADDR; i < NIADDR;
 		    i++, lbn += tmpval, tmpval *= NINDIR(fs)) {
 			/* Release a whole indirect tree. */
 			if (lbn > lastlbn) {
 				setup_freeindir(freeblks, ip, i, -lbn -i,
 				    needj);
 				continue;
 			}
 			iboff = i + NDADDR;
 			/*
 			 * Traverse partially truncated indirect tree.
 			 */
 			if (lbn <= lastlbn && lbn + tmpval - 1 > lastlbn)
 				setup_trunc_indir(freeblks, ip, -lbn - i,
 				    lastlbn, DIP(ip, i_ib[i]));
 		}
 		/*
 		 * Handle partial truncation to a frag boundary.
 		 */
 		if (frags) {
 			ufs2_daddr_t blkno;
 			long oldfrags;
 
 			oldfrags = blksize(fs, ip, lastlbn);
 			blkno = DIP(ip, i_db[lastlbn]);
 			if (blkno && oldfrags != frags) {
 				oldfrags -= frags;
 				oldfrags = numfrags(ip->i_fs, oldfrags);
 				blkno += numfrags(ip->i_fs, frags);
 				newfreework(ump, freeblks, NULL, lastlbn,
 				    blkno, oldfrags, 0, needj);
 				if (needj)
 					adjust_newfreework(freeblks,
 					    numfrags(ip->i_fs, frags));
 			} else if (blkno == 0)
 				allocblock = 1;
 		}
 		/*
 		 * Add a journal record for partial truncate if we are
 		 * handling indirect blocks.  Non-indirects need no extra
 		 * journaling.
 		 */
 		if (length != 0 && lastlbn >= NDADDR) {
 			ip->i_flag |= IN_TRUNCATED;
 			newjtrunc(freeblks, length, 0);
 		}
 		ip->i_size = length;
 		DIP_SET(ip, i_size, ip->i_size);
 		datablocks = DIP(ip, i_blocks) - extblocks;
 		if (length != 0)
 			datablocks = blkcount(ip->i_fs, datablocks, length);
 		freeblks->fb_len = length;
 	}
 	if ((flags & IO_EXT) != 0) {
 		for (i = 0; i < NXADDR; i++)
 			setup_freeext(freeblks, ip, i, needj);
 		ip->i_din2->di_extsize = 0;
 		datablocks += extblocks;
 	}
 #ifdef QUOTA
 	/* Reference the quotas in case the block count is wrong in the end. */
 	quotaref(vp, freeblks->fb_quota);
 	(void) chkdq(ip, -datablocks, NOCRED, 0);
 #endif
 	freeblks->fb_chkcnt = -datablocks;
 	UFS_LOCK(ump);
 	fs->fs_pendingblocks += datablocks;
 	UFS_UNLOCK(ump);
 	DIP_SET(ip, i_blocks, DIP(ip, i_blocks) - datablocks);
 	/*
 	 * Handle truncation of incomplete alloc direct dependencies.  We
 	 * hold the inode block locked to prevent incomplete dependencies
 	 * from reaching the disk while we are eliminating those that
 	 * have been truncated.  This is a partially inlined ffs_update().
 	 */
 	ufs_itimes(vp);
 	ip->i_flag &= ~(IN_LAZYACCESS | IN_LAZYMOD | IN_MODIFIED);
 	error = bread(ip->i_devvp, fsbtodb(fs, ino_to_fsba(fs, ip->i_number)),
 	    (int)fs->fs_bsize, cred, &bp);
 	if (error) {
 		brelse(bp);
 		softdep_error("softdep_journal_freeblocks", error);
 		return;
 	}
 	if (bp->b_bufsize == fs->fs_bsize)
 		bp->b_flags |= B_CLUSTEROK;
 	softdep_update_inodeblock(ip, bp, 0);
 	if (ump->um_fstype == UFS1)
 		*((struct ufs1_dinode *)bp->b_data +
 		    ino_to_fsbo(fs, ip->i_number)) = *ip->i_din1;
 	else
 		*((struct ufs2_dinode *)bp->b_data +
 		    ino_to_fsbo(fs, ip->i_number)) = *ip->i_din2;
 	ACQUIRE_LOCK(ump);
 	(void) inodedep_lookup(mp, ip->i_number, dflags, &inodedep);
 	if ((inodedep->id_state & IOSTARTED) != 0)
 		panic("softdep_setup_freeblocks: inode busy");
 	/*
 	 * Add the freeblks structure to the list of operations that
 	 * must await the zero'ed inode being written to disk. If we
 	 * still have a bitmap dependency (needj), then the inode
 	 * has never been written to disk, so we can process the
 	 * freeblks below once we have deleted the dependencies.
 	 */
 	if (needj)
 		WORKLIST_INSERT(&bp->b_dep, &freeblks->fb_list);
 	else
 		freeblks->fb_state |= COMPLETE;
 	if ((flags & IO_NORMAL) != 0) {
 		TAILQ_FOREACH_SAFE(adp, &inodedep->id_inoupdt, ad_next, adpn) {
 			if (adp->ad_offset > iboff)
 				cancel_allocdirect(&inodedep->id_inoupdt, adp,
 				    freeblks);
 			/*
 			 * Truncate the allocdirect.  We could eliminate
 			 * or modify journal records as well.
 			 */
 			else if (adp->ad_offset == iboff && frags)
 				adp->ad_newsize = frags;
 		}
 	}
 	if ((flags & IO_EXT) != 0)
 		while ((adp = TAILQ_FIRST(&inodedep->id_extupdt)) != 0)
 			cancel_allocdirect(&inodedep->id_extupdt, adp,
 			    freeblks);
 	/*
 	 * Scan the bufwait list for newblock dependencies that will never
 	 * make it to disk.
 	 */
 	LIST_FOREACH_SAFE(wk, &inodedep->id_bufwait, wk_list, wkn) {
 		if (wk->wk_type != D_ALLOCDIRECT)
 			continue;
 		adp = WK_ALLOCDIRECT(wk);
 		if (((flags & IO_NORMAL) != 0 && (adp->ad_offset > iboff)) ||
 		    ((flags & IO_EXT) != 0 && (adp->ad_state & EXTDATA))) {
 			cancel_jfreeblk(freeblks, adp->ad_newblkno);
 			cancel_newblk(WK_NEWBLK(wk), NULL, &freeblks->fb_jwork);
 			WORKLIST_INSERT(&freeblks->fb_freeworkhd, wk);
 		}
 	}
 	/*
 	 * Add journal work.
 	 */
 	LIST_FOREACH(jblkdep, &freeblks->fb_jblkdephd, jb_deps)
 		add_to_journal(&jblkdep->jb_list);
 	FREE_LOCK(ump);
 	bdwrite(bp);
 	/*
 	 * Truncate dependency structures beyond length.
 	 */
 	trunc_dependencies(ip, freeblks, lastlbn, frags, flags);
 	/*
 	 * This is only set when we need to allocate a fragment because
 	 * none existed at the end of a frag-sized file.  It handles only
 	 * allocating a new, zero filled block.
 	 */
 	if (allocblock) {
 		ip->i_size = length - lastoff;
 		DIP_SET(ip, i_size, ip->i_size);
 		error = UFS_BALLOC(vp, length - 1, 1, cred, BA_CLRBUF, &bp);
 		if (error != 0) {
 			softdep_error("softdep_journal_freeblks", error);
 			return;
 		}
 		ip->i_size = length;
 		DIP_SET(ip, i_size, length);
 		ip->i_flag |= IN_CHANGE | IN_UPDATE;
 		allocbuf(bp, frags);
 		ffs_update(vp, 0);
 		bawrite(bp);
 	} else if (lastoff != 0 && vp->v_type != VDIR) {
 		int size;
 
 		/*
 		 * Zero the end of a truncated frag or block.
 		 */
 		size = sblksize(fs, length, lastlbn);
 		error = bread(vp, lastlbn, size, cred, &bp);
 		if (error) {
 			softdep_error("softdep_journal_freeblks", error);
 			return;
 		}
 		bzero((char *)bp->b_data + lastoff, size - lastoff);
 		bawrite(bp);
 
 	}
 	ACQUIRE_LOCK(ump);
 	inodedep_lookup(mp, ip->i_number, dflags, &inodedep);
 	TAILQ_INSERT_TAIL(&inodedep->id_freeblklst, freeblks, fb_next);
 	freeblks->fb_state |= DEPCOMPLETE | ONDEPLIST;
 	/*
 	 * We zero earlier truncations so they don't erroneously
 	 * update i_blocks.
 	 */
 	if (freeblks->fb_len == 0 && (flags & IO_NORMAL) != 0)
 		TAILQ_FOREACH(fbn, &inodedep->id_freeblklst, fb_next)
 			fbn->fb_len = 0;
 	if ((freeblks->fb_state & ALLCOMPLETE) == ALLCOMPLETE &&
 	    LIST_EMPTY(&freeblks->fb_jblkdephd))
 		freeblks->fb_state |= INPROGRESS;
 	else
 		freeblks = NULL;
 	FREE_LOCK(ump);
 	if (freeblks)
 		handle_workitem_freeblocks(freeblks, 0);
 	trunc_pages(ip, length, extblocks, flags);
 
 }
 
 /*
  * Flush a JOP_SYNC to the journal.
  */
 void
 softdep_journal_fsync(ip)
 	struct inode *ip;
 {
 	struct jfsync *jfsync;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_journal_fsync called on non-softdep filesystem"));
 	if ((ip->i_flag & IN_TRUNCATED) == 0)
 		return;
 	ip->i_flag &= ~IN_TRUNCATED;
 	jfsync = malloc(sizeof(*jfsync), M_JFSYNC, M_SOFTDEP_FLAGS | M_ZERO);
 	workitem_alloc(&jfsync->jfs_list, D_JFSYNC, UFSTOVFS(ip->i_ump));
 	jfsync->jfs_size = ip->i_size;
 	jfsync->jfs_ino = ip->i_number;
 	ACQUIRE_LOCK(ip->i_ump);
 	add_to_journal(&jfsync->jfs_list);
 	jwait(&jfsync->jfs_list, MNT_WAIT);
 	FREE_LOCK(ip->i_ump);
 }
 
 /*
  * Block de-allocation dependencies.
  * 
  * When blocks are de-allocated, the on-disk pointers must be nullified before
  * the blocks are made available for use by other files.  (The true
  * requirement is that old pointers must be nullified before new on-disk
  * pointers are set.  We chose this slightly more stringent requirement to
  * reduce complexity.) Our implementation handles this dependency by updating
  * the inode (or indirect block) appropriately but delaying the actual block
  * de-allocation (i.e., freemap and free space count manipulation) until
  * after the updated versions reach stable storage.  After the disk is
  * updated, the blocks can be safely de-allocated whenever it is convenient.
  * This implementation handles only the common case of reducing a file's
  * length to zero. Other cases are handled by the conventional synchronous
  * write approach.
  *
  * The ffs implementation with which we worked double-checks
  * the state of the block pointers and file size as it reduces
  * a file's length.  Some of this code is replicated here in our
  * soft updates implementation.  The freeblks->fb_chkcnt field is
  * used to transfer a part of this information to the procedure
  * that eventually de-allocates the blocks.
  *
  * This routine should be called from the routine that shortens
  * a file's length, before the inode's size or block pointers
  * are modified. It will save the block pointer information for
  * later release and zero the inode so that the calling routine
  * can release it.
  */
 void
 softdep_setup_freeblocks(ip, length, flags)
 	struct inode *ip;	/* The inode whose length is to be reduced */
 	off_t length;		/* The new length for the file */
 	int flags;		/* IO_EXT and/or IO_NORMAL */
 {
 	struct ufs1_dinode *dp1;
 	struct ufs2_dinode *dp2;
 	struct freeblks *freeblks;
 	struct inodedep *inodedep;
 	struct allocdirect *adp;
 	struct ufsmount *ump;
 	struct buf *bp;
 	struct fs *fs;
 	ufs2_daddr_t extblocks, datablocks;
 	struct mount *mp;
 	int i, delay, error, dflags;
 	ufs_lbn_t tmpval;
 	ufs_lbn_t lbn;
 
 	ump = ip->i_ump;
 	mp = UFSTOVFS(ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_freeblocks called on non-softdep filesystem"));
 	CTR2(KTR_SUJ, "softdep_setup_freeblks: ip %d length %ld",
 	    ip->i_number, length);
 	KASSERT(length == 0, ("softdep_setup_freeblocks: non-zero length"));
 	fs = ip->i_fs;
 	freeblks = newfreeblks(mp, ip);
 	extblocks = 0;
 	datablocks = 0;
 	if (fs->fs_magic == FS_UFS2_MAGIC)
 		extblocks = btodb(fragroundup(fs, ip->i_din2->di_extsize));
 	if ((flags & IO_NORMAL) != 0) {
 		for (i = 0; i < NDADDR; i++)
 			setup_freedirect(freeblks, ip, i, 0);
 		for (i = 0, tmpval = NINDIR(fs), lbn = NDADDR; i < NIADDR;
 		    i++, lbn += tmpval, tmpval *= NINDIR(fs))
 			setup_freeindir(freeblks, ip, i, -lbn -i, 0);
 		ip->i_size = 0;
 		DIP_SET(ip, i_size, 0);
 		datablocks = DIP(ip, i_blocks) - extblocks;
 	}
 	if ((flags & IO_EXT) != 0) {
 		for (i = 0; i < NXADDR; i++)
 			setup_freeext(freeblks, ip, i, 0);
 		ip->i_din2->di_extsize = 0;
 		datablocks += extblocks;
 	}
 #ifdef QUOTA
 	/* Reference the quotas in case the block count is wrong in the end. */
 	quotaref(ITOV(ip), freeblks->fb_quota);
 	(void) chkdq(ip, -datablocks, NOCRED, 0);
 #endif
 	freeblks->fb_chkcnt = -datablocks;
 	UFS_LOCK(ump);
 	fs->fs_pendingblocks += datablocks;
 	UFS_UNLOCK(ump);
 	DIP_SET(ip, i_blocks, DIP(ip, i_blocks) - datablocks);
 	/*
 	 * Push the zero'ed inode to to its disk buffer so that we are free
 	 * to delete its dependencies below. Once the dependencies are gone
 	 * the buffer can be safely released.
 	 */
 	if ((error = bread(ip->i_devvp,
 	    fsbtodb(fs, ino_to_fsba(fs, ip->i_number)),
 	    (int)fs->fs_bsize, NOCRED, &bp)) != 0) {
 		brelse(bp);
 		softdep_error("softdep_setup_freeblocks", error);
 	}
 	if (ump->um_fstype == UFS1) {
 		dp1 = ((struct ufs1_dinode *)bp->b_data +
 		    ino_to_fsbo(fs, ip->i_number));
 		ip->i_din1->di_freelink = dp1->di_freelink;
 		*dp1 = *ip->i_din1;
 	} else {
 		dp2 = ((struct ufs2_dinode *)bp->b_data +
 		    ino_to_fsbo(fs, ip->i_number));
 		ip->i_din2->di_freelink = dp2->di_freelink;
 		*dp2 = *ip->i_din2;
 	}
 	/*
 	 * Find and eliminate any inode dependencies.
 	 */
 	ACQUIRE_LOCK(ump);
 	dflags = DEPALLOC;
 	if (IS_SNAPSHOT(ip))
 		dflags |= NODELAY;
 	(void) inodedep_lookup(mp, ip->i_number, dflags, &inodedep);
 	if ((inodedep->id_state & IOSTARTED) != 0)
 		panic("softdep_setup_freeblocks: inode busy");
 	/*
 	 * Add the freeblks structure to the list of operations that
 	 * must await the zero'ed inode being written to disk. If we
 	 * still have a bitmap dependency (delay == 0), then the inode
 	 * has never been written to disk, so we can process the
 	 * freeblks below once we have deleted the dependencies.
 	 */
 	delay = (inodedep->id_state & DEPCOMPLETE);
 	if (delay)
 		WORKLIST_INSERT(&bp->b_dep, &freeblks->fb_list);
 	else
 		freeblks->fb_state |= COMPLETE;
 	/*
 	 * Because the file length has been truncated to zero, any
 	 * pending block allocation dependency structures associated
 	 * with this inode are obsolete and can simply be de-allocated.
 	 * We must first merge the two dependency lists to get rid of
 	 * any duplicate freefrag structures, then purge the merged list.
 	 * If we still have a bitmap dependency, then the inode has never
 	 * been written to disk, so we can free any fragments without delay.
 	 */
 	if (flags & IO_NORMAL) {
 		merge_inode_lists(&inodedep->id_newinoupdt,
 		    &inodedep->id_inoupdt);
 		while ((adp = TAILQ_FIRST(&inodedep->id_inoupdt)) != 0)
 			cancel_allocdirect(&inodedep->id_inoupdt, adp,
 			    freeblks);
 	}
 	if (flags & IO_EXT) {
 		merge_inode_lists(&inodedep->id_newextupdt,
 		    &inodedep->id_extupdt);
 		while ((adp = TAILQ_FIRST(&inodedep->id_extupdt)) != 0)
 			cancel_allocdirect(&inodedep->id_extupdt, adp,
 			    freeblks);
 	}
 	FREE_LOCK(ump);
 	bdwrite(bp);
 	trunc_dependencies(ip, freeblks, -1, 0, flags);
 	ACQUIRE_LOCK(ump);
 	if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) != 0)
 		(void) free_inodedep(inodedep);
 	freeblks->fb_state |= DEPCOMPLETE;
 	/*
 	 * If the inode with zeroed block pointers is now on disk
 	 * we can start freeing blocks.
 	 */  
 	if ((freeblks->fb_state & ALLCOMPLETE) == ALLCOMPLETE)
 		freeblks->fb_state |= INPROGRESS;
 	else
 		freeblks = NULL;
 	FREE_LOCK(ump);
 	if (freeblks)
 		handle_workitem_freeblocks(freeblks, 0);
 	trunc_pages(ip, length, extblocks, flags);
 }
 
 /*
  * Eliminate pages from the page cache that back parts of this inode and
  * adjust the vnode pager's idea of our size.  This prevents stale data
  * from hanging around in the page cache.
  */
 static void
 trunc_pages(ip, length, extblocks, flags)
 	struct inode *ip;
 	off_t length;
 	ufs2_daddr_t extblocks;
 	int flags;
 {
 	struct vnode *vp;
 	struct fs *fs;
 	ufs_lbn_t lbn;
 	off_t end, extend;
 
 	vp = ITOV(ip);
 	fs = ip->i_fs;
 	extend = OFF_TO_IDX(lblktosize(fs, -extblocks));
 	if ((flags & IO_EXT) != 0)
 		vn_pages_remove(vp, extend, 0);
 	if ((flags & IO_NORMAL) == 0)
 		return;
 	BO_LOCK(&vp->v_bufobj);
 	drain_output(vp);
 	BO_UNLOCK(&vp->v_bufobj);
 	/*
 	 * The vnode pager eliminates file pages we eliminate indirects
 	 * below.
 	 */
 	vnode_pager_setsize(vp, length);
 	/*
 	 * Calculate the end based on the last indirect we want to keep.  If
 	 * the block extends into indirects we can just use the negative of
 	 * its lbn.  Doubles and triples exist at lower numbers so we must
 	 * be careful not to remove those, if they exist.  double and triple
 	 * indirect lbns do not overlap with others so it is not important
 	 * to verify how many levels are required.
 	 */
 	lbn = lblkno(fs, length);
 	if (lbn >= NDADDR) {
 		/* Calculate the virtual lbn of the triple indirect. */
 		lbn = -lbn - (NIADDR - 1);
 		end = OFF_TO_IDX(lblktosize(fs, lbn));
 	} else
 		end = extend;
 	vn_pages_remove(vp, OFF_TO_IDX(OFF_MAX), end);
 }
 
 /*
  * See if the buf bp is in the range eliminated by truncation.
  */
 static int
 trunc_check_buf(bp, blkoffp, lastlbn, lastoff, flags)
 	struct buf *bp;
 	int *blkoffp;
 	ufs_lbn_t lastlbn;
 	int lastoff;
 	int flags;
 {
 	ufs_lbn_t lbn;
 
 	*blkoffp = 0;
 	/* Only match ext/normal blocks as appropriate. */
 	if (((flags & IO_EXT) == 0 && (bp->b_xflags & BX_ALTDATA)) ||
 	    ((flags & IO_NORMAL) == 0 && (bp->b_xflags & BX_ALTDATA) == 0))
 		return (0);
 	/* ALTDATA is always a full truncation. */
 	if ((bp->b_xflags & BX_ALTDATA) != 0)
 		return (1);
 	/* -1 is full truncation. */
 	if (lastlbn == -1)
 		return (1);
 	/*
 	 * If this is a partial truncate we only want those
 	 * blocks and indirect blocks that cover the range
 	 * we're after.
 	 */
 	lbn = bp->b_lblkno;
 	if (lbn < 0)
 		lbn = -(lbn + lbn_level(lbn));
 	if (lbn < lastlbn)
 		return (0);
 	/* Here we only truncate lblkno if it's partial. */
 	if (lbn == lastlbn) {
 		if (lastoff == 0)
 			return (0);
 		*blkoffp = lastoff;
 	}
 	return (1);
 }
 
 /*
  * Eliminate any dependencies that exist in memory beyond lblkno:off
  */
 static void
 trunc_dependencies(ip, freeblks, lastlbn, lastoff, flags)
 	struct inode *ip;
 	struct freeblks *freeblks;
 	ufs_lbn_t lastlbn;
 	int lastoff;
 	int flags;
 {
 	struct bufobj *bo;
 	struct vnode *vp;
 	struct buf *bp;
 	struct fs *fs;
 	int blkoff;
 
 	/*
 	 * We must wait for any I/O in progress to finish so that
 	 * all potential buffers on the dirty list will be visible.
 	 * Once they are all there, walk the list and get rid of
 	 * any dependencies.
 	 */
 	fs = ip->i_fs;
 	vp = ITOV(ip);
 	bo = &vp->v_bufobj;
 	BO_LOCK(bo);
 	drain_output(vp);
 	TAILQ_FOREACH(bp, &bo->bo_dirty.bv_hd, b_bobufs)
 		bp->b_vflags &= ~BV_SCANNED;
 restart:
 	TAILQ_FOREACH(bp, &bo->bo_dirty.bv_hd, b_bobufs) {
 		if (bp->b_vflags & BV_SCANNED)
 			continue;
 		if (!trunc_check_buf(bp, &blkoff, lastlbn, lastoff, flags)) {
 			bp->b_vflags |= BV_SCANNED;
 			continue;
 		}
 		KASSERT(bp->b_bufobj == bo, ("Wrong object in buffer"));
 		if ((bp = getdirtybuf(bp, BO_LOCKPTR(bo), MNT_WAIT)) == NULL)
 			goto restart;
 		BO_UNLOCK(bo);
 		if (deallocate_dependencies(bp, freeblks, blkoff))
 			bqrelse(bp);
 		else
 			brelse(bp);
 		BO_LOCK(bo);
 		goto restart;
 	}
 	/*
 	 * Now do the work of vtruncbuf while also matching indirect blocks.
 	 */
 	TAILQ_FOREACH(bp, &bo->bo_clean.bv_hd, b_bobufs)
 		bp->b_vflags &= ~BV_SCANNED;
 cleanrestart:
 	TAILQ_FOREACH(bp, &bo->bo_clean.bv_hd, b_bobufs) {
 		if (bp->b_vflags & BV_SCANNED)
 			continue;
 		if (!trunc_check_buf(bp, &blkoff, lastlbn, lastoff, flags)) {
 			bp->b_vflags |= BV_SCANNED;
 			continue;
 		}
 		if (BUF_LOCK(bp,
 		    LK_EXCLUSIVE | LK_SLEEPFAIL | LK_INTERLOCK,
 		    BO_LOCKPTR(bo)) == ENOLCK) {
 			BO_LOCK(bo);
 			goto cleanrestart;
 		}
 		bp->b_vflags |= BV_SCANNED;
 		bremfree(bp);
 		if (blkoff != 0) {
 			allocbuf(bp, blkoff);
 			bqrelse(bp);
 		} else {
 			bp->b_flags |= B_INVAL | B_NOCACHE | B_RELBUF;
 			brelse(bp);
 		}
 		BO_LOCK(bo);
 		goto cleanrestart;
 	}
 	drain_output(vp);
 	BO_UNLOCK(bo);
 }
 
 static int
 cancel_pagedep(pagedep, freeblks, blkoff)
 	struct pagedep *pagedep;
 	struct freeblks *freeblks;
 	int blkoff;
 {
 	struct jremref *jremref;
 	struct jmvref *jmvref;
 	struct dirrem *dirrem, *tmp;
 	int i;
 
 	/*
 	 * Copy any directory remove dependencies to the list
 	 * to be processed after the freeblks proceeds.  If
 	 * directory entry never made it to disk they
 	 * can be dumped directly onto the work list.
 	 */
 	LIST_FOREACH_SAFE(dirrem, &pagedep->pd_dirremhd, dm_next, tmp) {
 		/* Skip this directory removal if it is intended to remain. */
 		if (dirrem->dm_offset < blkoff)
 			continue;
 		/*
 		 * If there are any dirrems we wait for the journal write
 		 * to complete and then restart the buf scan as the lock
 		 * has been dropped.
 		 */
 		while ((jremref = LIST_FIRST(&dirrem->dm_jremrefhd)) != NULL) {
 			jwait(&jremref->jr_list, MNT_WAIT);
 			return (ERESTART);
 		}
 		LIST_REMOVE(dirrem, dm_next);
 		dirrem->dm_dirinum = pagedep->pd_ino;
 		WORKLIST_INSERT(&freeblks->fb_freeworkhd, &dirrem->dm_list);
 	}
 	while ((jmvref = LIST_FIRST(&pagedep->pd_jmvrefhd)) != NULL) {
 		jwait(&jmvref->jm_list, MNT_WAIT);
 		return (ERESTART);
 	}
 	/*
 	 * When we're partially truncating a pagedep we just want to flush
 	 * journal entries and return.  There can not be any adds in the
 	 * truncated portion of the directory and newblk must remain if
 	 * part of the block remains.
 	 */
 	if (blkoff != 0) {
 		struct diradd *dap;
 
 		LIST_FOREACH(dap, &pagedep->pd_pendinghd, da_pdlist)
 			if (dap->da_offset > blkoff)
 				panic("cancel_pagedep: diradd %p off %d > %d",
 				    dap, dap->da_offset, blkoff);
 		for (i = 0; i < DAHASHSZ; i++)
 			LIST_FOREACH(dap, &pagedep->pd_diraddhd[i], da_pdlist)
 				if (dap->da_offset > blkoff)
 					panic("cancel_pagedep: diradd %p off %d > %d",
 					    dap, dap->da_offset, blkoff);
 		return (0);
 	}
 	/*
 	 * There should be no directory add dependencies present
 	 * as the directory could not be truncated until all
 	 * children were removed.
 	 */
 	KASSERT(LIST_FIRST(&pagedep->pd_pendinghd) == NULL,
 	    ("deallocate_dependencies: pendinghd != NULL"));
 	for (i = 0; i < DAHASHSZ; i++)
 		KASSERT(LIST_FIRST(&pagedep->pd_diraddhd[i]) == NULL,
 		    ("deallocate_dependencies: diraddhd != NULL"));
 	if ((pagedep->pd_state & NEWBLOCK) != 0)
 		free_newdirblk(pagedep->pd_newdirblk);
 	if (free_pagedep(pagedep) == 0)
 		panic("Failed to free pagedep %p", pagedep);
 	return (0);
 }
 
 /*
  * Reclaim any dependency structures from a buffer that is about to
  * be reallocated to a new vnode. The buffer must be locked, thus,
  * no I/O completion operations can occur while we are manipulating
  * its associated dependencies. The mutex is held so that other I/O's
  * associated with related dependencies do not occur.
  */
 static int
 deallocate_dependencies(bp, freeblks, off)
 	struct buf *bp;
 	struct freeblks *freeblks;
 	int off;
 {
 	struct indirdep *indirdep;
 	struct pagedep *pagedep;
 	struct allocdirect *adp;
 	struct worklist *wk, *wkn;
 	struct ufsmount *ump;
 
 	if ((wk = LIST_FIRST(&bp->b_dep)) == NULL)
 		goto done;
 	ump = VFSTOUFS(wk->wk_mp);
 	ACQUIRE_LOCK(ump);
 	LIST_FOREACH_SAFE(wk, &bp->b_dep, wk_list, wkn) {
 		switch (wk->wk_type) {
 		case D_INDIRDEP:
 			indirdep = WK_INDIRDEP(wk);
 			if (bp->b_lblkno >= 0 ||
 			    bp->b_blkno != indirdep->ir_savebp->b_lblkno)
 				panic("deallocate_dependencies: not indir");
 			cancel_indirdep(indirdep, bp, freeblks);
 			continue;
 
 		case D_PAGEDEP:
 			pagedep = WK_PAGEDEP(wk);
 			if (cancel_pagedep(pagedep, freeblks, off)) {
 				FREE_LOCK(ump);
 				return (ERESTART);
 			}
 			continue;
 
 		case D_ALLOCINDIR:
 			/*
 			 * Simply remove the allocindir, we'll find it via
 			 * the indirdep where we can clear pointers if
 			 * needed.
 			 */
 			WORKLIST_REMOVE(wk);
 			continue;
 
 		case D_FREEWORK:
 			/*
 			 * A truncation is waiting for the zero'd pointers
 			 * to be written.  It can be freed when the freeblks
 			 * is journaled.
 			 */
 			WORKLIST_REMOVE(wk);
 			wk->wk_state |= ONDEPLIST;
 			WORKLIST_INSERT(&freeblks->fb_freeworkhd, wk);
 			break;
 
 		case D_ALLOCDIRECT:
 			adp = WK_ALLOCDIRECT(wk);
 			if (off != 0)
 				continue;
 			/* FALLTHROUGH */
 		default:
 			panic("deallocate_dependencies: Unexpected type %s",
 			    TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 	}
 	FREE_LOCK(ump);
 done:
 	/*
 	 * Don't throw away this buf, we were partially truncating and
 	 * some deps may always remain.
 	 */
 	if (off) {
 		allocbuf(bp, off);
 		bp->b_vflags |= BV_SCANNED;
 		return (EBUSY);
 	}
 	bp->b_flags |= B_INVAL | B_NOCACHE;
 
 	return (0);
 }
 
 /*
  * An allocdirect is being canceled due to a truncate.  We must make sure
  * the journal entry is released in concert with the blkfree that releases
  * the storage.  Completed journal entries must not be released until the
  * space is no longer pointed to by the inode or in the bitmap.
  */
 static void
 cancel_allocdirect(adphead, adp, freeblks)
 	struct allocdirectlst *adphead;
 	struct allocdirect *adp;
 	struct freeblks *freeblks;
 {
 	struct freework *freework;
 	struct newblk *newblk;
 	struct worklist *wk;
 
 	TAILQ_REMOVE(adphead, adp, ad_next);
 	newblk = (struct newblk *)adp;
 	freework = NULL;
 	/*
 	 * Find the correct freework structure.
 	 */
 	LIST_FOREACH(wk, &freeblks->fb_freeworkhd, wk_list) {
 		if (wk->wk_type != D_FREEWORK)
 			continue;
 		freework = WK_FREEWORK(wk);
 		if (freework->fw_blkno == newblk->nb_newblkno)
 			break;
 	}
 	if (freework == NULL)
 		panic("cancel_allocdirect: Freework not found");
 	/*
 	 * If a newblk exists at all we still have the journal entry that
 	 * initiated the allocation so we do not need to journal the free.
 	 */
 	cancel_jfreeblk(freeblks, freework->fw_blkno);
 	/*
 	 * If the journal hasn't been written the jnewblk must be passed
 	 * to the call to ffs_blkfree that reclaims the space.  We accomplish
 	 * this by linking the journal dependency into the freework to be
 	 * freed when freework_freeblock() is called.  If the journal has
 	 * been written we can simply reclaim the journal space when the
 	 * freeblks work is complete.
 	 */
 	freework->fw_jnewblk = cancel_newblk(newblk, &freework->fw_list,
 	    &freeblks->fb_jwork);
 	WORKLIST_INSERT(&freeblks->fb_freeworkhd, &newblk->nb_list);
 }
 
 
 /*
  * Cancel a new block allocation.  May be an indirect or direct block.  We
  * remove it from various lists and return any journal record that needs to
  * be resolved by the caller.
  *
  * A special consideration is made for indirects which were never pointed
  * at on disk and will never be found once this block is released.
  */
 static struct jnewblk *
 cancel_newblk(newblk, wk, wkhd)
 	struct newblk *newblk;
 	struct worklist *wk;
 	struct workhead *wkhd;
 {
 	struct jnewblk *jnewblk;
 
 	CTR1(KTR_SUJ, "cancel_newblk: blkno %jd", newblk->nb_newblkno);
 	    
 	newblk->nb_state |= GOINGAWAY;
 	/*
 	 * Previously we traversed the completedhd on each indirdep
 	 * attached to this newblk to cancel them and gather journal
 	 * work.  Since we need only the oldest journal segment and
 	 * the lowest point on the tree will always have the oldest
 	 * journal segment we are free to release the segments
 	 * of any subordinates and may leave the indirdep list to
 	 * indirdep_complete() when this newblk is freed.
 	 */
 	if (newblk->nb_state & ONDEPLIST) {
 		newblk->nb_state &= ~ONDEPLIST;
 		LIST_REMOVE(newblk, nb_deps);
 	}
 	if (newblk->nb_state & ONWORKLIST)
 		WORKLIST_REMOVE(&newblk->nb_list);
 	/*
 	 * If the journal entry hasn't been written we save a pointer to
 	 * the dependency that frees it until it is written or the
 	 * superseding operation completes.
 	 */
 	jnewblk = newblk->nb_jnewblk;
 	if (jnewblk != NULL && wk != NULL) {
 		newblk->nb_jnewblk = NULL;
 		jnewblk->jn_dep = wk;
 	}
 	if (!LIST_EMPTY(&newblk->nb_jwork))
 		jwork_move(wkhd, &newblk->nb_jwork);
 	/*
 	 * When truncating we must free the newdirblk early to remove
 	 * the pagedep from the hash before returning.
 	 */
 	if ((wk = LIST_FIRST(&newblk->nb_newdirblk)) != NULL)
 		free_newdirblk(WK_NEWDIRBLK(wk));
 	if (!LIST_EMPTY(&newblk->nb_newdirblk))
 		panic("cancel_newblk: extra newdirblk");
 
 	return (jnewblk);
 }
 
 /*
  * Schedule the freefrag associated with a newblk to be released once
  * the pointers are written and the previous block is no longer needed.
  */
 static void
 newblk_freefrag(newblk)
 	struct newblk *newblk;
 {
 	struct freefrag *freefrag;
 
 	if (newblk->nb_freefrag == NULL)
 		return;
 	freefrag = newblk->nb_freefrag;
 	newblk->nb_freefrag = NULL;
 	freefrag->ff_state |= COMPLETE;
 	if ((freefrag->ff_state & ALLCOMPLETE) == ALLCOMPLETE)
 		add_to_worklist(&freefrag->ff_list, 0);
 }
 
 /*
  * Free a newblk. Generate a new freefrag work request if appropriate.
  * This must be called after the inode pointer and any direct block pointers
  * are valid or fully removed via truncate or frag extension.
  */
 static void
 free_newblk(newblk)
 	struct newblk *newblk;
 {
 	struct indirdep *indirdep;
 	struct worklist *wk;
 
 	KASSERT(newblk->nb_jnewblk == NULL,
 	    ("free_newblk: jnewblk %p still attached", newblk->nb_jnewblk));
 	KASSERT(newblk->nb_list.wk_type != D_NEWBLK,
 	    ("free_newblk: unclaimed newblk"));
 	LOCK_OWNED(VFSTOUFS(newblk->nb_list.wk_mp));
 	newblk_freefrag(newblk);
 	if (newblk->nb_state & ONDEPLIST)
 		LIST_REMOVE(newblk, nb_deps);
 	if (newblk->nb_state & ONWORKLIST)
 		WORKLIST_REMOVE(&newblk->nb_list);
 	LIST_REMOVE(newblk, nb_hash);
 	if ((wk = LIST_FIRST(&newblk->nb_newdirblk)) != NULL)
 		free_newdirblk(WK_NEWDIRBLK(wk));
 	if (!LIST_EMPTY(&newblk->nb_newdirblk))
 		panic("free_newblk: extra newdirblk");
 	while ((indirdep = LIST_FIRST(&newblk->nb_indirdeps)) != NULL)
 		indirdep_complete(indirdep);
 	handle_jwork(&newblk->nb_jwork);
 	WORKITEM_FREE(newblk, D_NEWBLK);
 }
 
 /*
  * Free a newdirblk. Clear the NEWBLOCK flag on its associated pagedep.
  * This routine must be called with splbio interrupts blocked.
  */
 static void
 free_newdirblk(newdirblk)
 	struct newdirblk *newdirblk;
 {
 	struct pagedep *pagedep;
 	struct diradd *dap;
 	struct worklist *wk;
 
 	LOCK_OWNED(VFSTOUFS(newdirblk->db_list.wk_mp));
 	WORKLIST_REMOVE(&newdirblk->db_list);
 	/*
 	 * If the pagedep is still linked onto the directory buffer
 	 * dependency chain, then some of the entries on the
 	 * pd_pendinghd list may not be committed to disk yet. In
 	 * this case, we will simply clear the NEWBLOCK flag and
 	 * let the pd_pendinghd list be processed when the pagedep
 	 * is next written. If the pagedep is no longer on the buffer
 	 * dependency chain, then all the entries on the pd_pending
 	 * list are committed to disk and we can free them here.
 	 */
 	pagedep = newdirblk->db_pagedep;
 	pagedep->pd_state &= ~NEWBLOCK;
 	if ((pagedep->pd_state & ONWORKLIST) == 0) {
 		while ((dap = LIST_FIRST(&pagedep->pd_pendinghd)) != NULL)
 			free_diradd(dap, NULL);
 		/*
 		 * If no dependencies remain, the pagedep will be freed.
 		 */
 		free_pagedep(pagedep);
 	}
 	/* Should only ever be one item in the list. */
 	while ((wk = LIST_FIRST(&newdirblk->db_mkdir)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		handle_written_mkdir(WK_MKDIR(wk), MKDIR_BODY);
 	}
 	WORKITEM_FREE(newdirblk, D_NEWDIRBLK);
 }
 
 /*
  * Prepare an inode to be freed. The actual free operation is not
  * done until the zero'ed inode has been written to disk.
  */
 void
 softdep_freefile(pvp, ino, mode)
 	struct vnode *pvp;
 	ino_t ino;
 	int mode;
 {
 	struct inode *ip = VTOI(pvp);
 	struct inodedep *inodedep;
 	struct freefile *freefile;
 	struct freeblks *freeblks;
 	struct ufsmount *ump;
 
 	ump = ip->i_ump;
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ump)) != 0,
 	    ("softdep_freefile called on non-softdep filesystem"));
 	/*
 	 * This sets up the inode de-allocation dependency.
 	 */
 	freefile = malloc(sizeof(struct freefile),
 		M_FREEFILE, M_SOFTDEP_FLAGS);
 	workitem_alloc(&freefile->fx_list, D_FREEFILE, pvp->v_mount);
 	freefile->fx_mode = mode;
 	freefile->fx_oldinum = ino;
 	freefile->fx_devvp = ip->i_devvp;
 	LIST_INIT(&freefile->fx_jwork);
 	UFS_LOCK(ump);
 	ip->i_fs->fs_pendinginodes += 1;
 	UFS_UNLOCK(ump);
 
 	/*
 	 * If the inodedep does not exist, then the zero'ed inode has
 	 * been written to disk. If the allocated inode has never been
 	 * written to disk, then the on-disk inode is zero'ed. In either
 	 * case we can free the file immediately.  If the journal was
 	 * canceled before being written the inode will never make it to
 	 * disk and we must send the canceled journal entrys to
 	 * ffs_freefile() to be cleared in conjunction with the bitmap.
 	 * Any blocks waiting on the inode to write can be safely freed
 	 * here as it will never been written.
 	 */
 	ACQUIRE_LOCK(ump);
 	inodedep_lookup(pvp->v_mount, ino, 0, &inodedep);
 	if (inodedep) {
 		/*
 		 * Clear out freeblks that no longer need to reference
 		 * this inode.
 		 */
 		while ((freeblks =
 		    TAILQ_FIRST(&inodedep->id_freeblklst)) != NULL) {
 			TAILQ_REMOVE(&inodedep->id_freeblklst, freeblks,
 			    fb_next);
 			freeblks->fb_state &= ~ONDEPLIST;
 		}
 		/*
 		 * Remove this inode from the unlinked list.
 		 */
 		if (inodedep->id_state & UNLINKED) {
 			/*
 			 * Save the journal work to be freed with the bitmap
 			 * before we clear UNLINKED.  Otherwise it can be lost
 			 * if the inode block is written.
 			 */
 			handle_bufwait(inodedep, &freefile->fx_jwork);
 			clear_unlinked_inodedep(inodedep);
 			/*
 			 * Re-acquire inodedep as we've dropped the
 			 * per-filesystem lock in clear_unlinked_inodedep().
 			 */
 			inodedep_lookup(pvp->v_mount, ino, 0, &inodedep);
 		}
 	}
 	if (inodedep == NULL || check_inode_unwritten(inodedep)) {
 		FREE_LOCK(ump);
 		handle_workitem_freefile(freefile);
 		return;
 	}
 	if ((inodedep->id_state & DEPCOMPLETE) == 0)
 		inodedep->id_state |= GOINGAWAY;
 	WORKLIST_INSERT(&inodedep->id_inowait, &freefile->fx_list);
 	FREE_LOCK(ump);
 	if (ip->i_number == ino)
 		ip->i_flag |= IN_MODIFIED;
 }
 
 /*
  * Check to see if an inode has never been written to disk. If
  * so free the inodedep and return success, otherwise return failure.
  * This routine must be called with splbio interrupts blocked.
  *
  * If we still have a bitmap dependency, then the inode has never
  * been written to disk. Drop the dependency as it is no longer
  * necessary since the inode is being deallocated. We set the
  * ALLCOMPLETE flags since the bitmap now properly shows that the
  * inode is not allocated. Even if the inode is actively being
  * written, it has been rolled back to its zero'ed state, so we
  * are ensured that a zero inode is what is on the disk. For short
  * lived files, this change will usually result in removing all the
  * dependencies from the inode so that it can be freed immediately.
  */
 static int
 check_inode_unwritten(inodedep)
 	struct inodedep *inodedep;
 {
 
 	LOCK_OWNED(VFSTOUFS(inodedep->id_list.wk_mp));
 
 	if ((inodedep->id_state & (DEPCOMPLETE | UNLINKED)) != 0 ||
 	    !LIST_EMPTY(&inodedep->id_dirremhd) ||
 	    !LIST_EMPTY(&inodedep->id_pendinghd) ||
 	    !LIST_EMPTY(&inodedep->id_bufwait) ||
 	    !LIST_EMPTY(&inodedep->id_inowait) ||
 	    !TAILQ_EMPTY(&inodedep->id_inoreflst) ||
 	    !TAILQ_EMPTY(&inodedep->id_inoupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_newinoupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_extupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_newextupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_freeblklst) ||
 	    inodedep->id_mkdiradd != NULL || 
 	    inodedep->id_nlinkdelta != 0)
 		return (0);
 	/*
 	 * Another process might be in initiate_write_inodeblock_ufs[12]
 	 * trying to allocate memory without holding "Softdep Lock".
 	 */
 	if ((inodedep->id_state & IOSTARTED) != 0 &&
 	    inodedep->id_savedino1 == NULL)
 		return (0);
 
 	if (inodedep->id_state & ONDEPLIST)
 		LIST_REMOVE(inodedep, id_deps);
 	inodedep->id_state &= ~ONDEPLIST;
 	inodedep->id_state |= ALLCOMPLETE;
 	inodedep->id_bmsafemap = NULL;
 	if (inodedep->id_state & ONWORKLIST)
 		WORKLIST_REMOVE(&inodedep->id_list);
 	if (inodedep->id_savedino1 != NULL) {
 		free(inodedep->id_savedino1, M_SAVEDINO);
 		inodedep->id_savedino1 = NULL;
 	}
 	if (free_inodedep(inodedep) == 0)
 		panic("check_inode_unwritten: busy inode");
 	return (1);
 }
 
 /*
  * Try to free an inodedep structure. Return 1 if it could be freed.
  */
 static int
 free_inodedep(inodedep)
 	struct inodedep *inodedep;
 {
 
 	LOCK_OWNED(VFSTOUFS(inodedep->id_list.wk_mp));
 	if ((inodedep->id_state & (ONWORKLIST | UNLINKED)) != 0 ||
 	    (inodedep->id_state & ALLCOMPLETE) != ALLCOMPLETE ||
 	    !LIST_EMPTY(&inodedep->id_dirremhd) ||
 	    !LIST_EMPTY(&inodedep->id_pendinghd) ||
 	    !LIST_EMPTY(&inodedep->id_bufwait) ||
 	    !LIST_EMPTY(&inodedep->id_inowait) ||
 	    !TAILQ_EMPTY(&inodedep->id_inoreflst) ||
 	    !TAILQ_EMPTY(&inodedep->id_inoupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_newinoupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_extupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_newextupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_freeblklst) ||
 	    inodedep->id_mkdiradd != NULL ||
 	    inodedep->id_nlinkdelta != 0 ||
 	    inodedep->id_savedino1 != NULL)
 		return (0);
 	if (inodedep->id_state & ONDEPLIST)
 		LIST_REMOVE(inodedep, id_deps);
 	LIST_REMOVE(inodedep, id_hash);
 	WORKITEM_FREE(inodedep, D_INODEDEP);
 	return (1);
 }
 
 /*
  * Free the block referenced by a freework structure.  The parent freeblks
  * structure is released and completed when the final cg bitmap reaches
  * the disk.  This routine may be freeing a jnewblk which never made it to
  * disk in which case we do not have to wait as the operation is undone
  * in memory immediately.
  */
 static void
 freework_freeblock(freework)
 	struct freework *freework;
 {
 	struct freeblks *freeblks;
 	struct jnewblk *jnewblk;
 	struct ufsmount *ump;
 	struct workhead wkhd;
 	struct fs *fs;
 	int bsize;
 	int needj;
 
 	ump = VFSTOUFS(freework->fw_list.wk_mp);
 	LOCK_OWNED(ump);
 	/*
 	 * Handle partial truncate separately.
 	 */
 	if (freework->fw_indir) {
 		complete_trunc_indir(freework);
 		return;
 	}
 	freeblks = freework->fw_freeblks;
 	fs = ump->um_fs;
 	needj = MOUNTEDSUJ(freeblks->fb_list.wk_mp) != 0;
 	bsize = lfragtosize(fs, freework->fw_frags);
 	LIST_INIT(&wkhd);
 	/*
 	 * DEPCOMPLETE is cleared in indirblk_insert() if the block lives
 	 * on the indirblk hashtable and prevents premature freeing.
 	 */
 	freework->fw_state |= DEPCOMPLETE;
 	/*
 	 * SUJ needs to wait for the segment referencing freed indirect
 	 * blocks to expire so that we know the checker will not confuse
 	 * a re-allocated indirect block with its old contents.
 	 */
 	if (needj && freework->fw_lbn <= -NDADDR)
 		indirblk_insert(freework);
 	/*
 	 * If we are canceling an existing jnewblk pass it to the free
 	 * routine, otherwise pass the freeblk which will ultimately
 	 * release the freeblks.  If we're not journaling, we can just
 	 * free the freeblks immediately.
 	 */
 	jnewblk = freework->fw_jnewblk;
 	if (jnewblk != NULL) {
 		cancel_jnewblk(jnewblk, &wkhd);
 		needj = 0;
 	} else if (needj) {
 		freework->fw_state |= DELAYEDFREE;
 		freeblks->fb_cgwait++;
 		WORKLIST_INSERT(&wkhd, &freework->fw_list);
 	}
 	FREE_LOCK(ump);
 	freeblks_free(ump, freeblks, btodb(bsize));
 	CTR4(KTR_SUJ,
 	    "freework_freeblock: ino %d blkno %jd lbn %jd size %ld",
 	    freeblks->fb_inum, freework->fw_blkno, freework->fw_lbn, bsize);
 	ffs_blkfree(ump, fs, freeblks->fb_devvp, freework->fw_blkno, bsize,
 	    freeblks->fb_inum, freeblks->fb_vtype, &wkhd);
 	ACQUIRE_LOCK(ump);
 	/*
 	 * The jnewblk will be discarded and the bits in the map never
 	 * made it to disk.  We can immediately free the freeblk.
 	 */
 	if (needj == 0)
 		handle_written_freework(freework);
 }
 
 /*
  * We enqueue freework items that need processing back on the freeblks and
  * add the freeblks to the worklist.  This makes it easier to find all work
  * required to flush a truncation in process_truncates().
  */
 static void
 freework_enqueue(freework)
 	struct freework *freework;
 {
 	struct freeblks *freeblks;
 
 	freeblks = freework->fw_freeblks;
 	if ((freework->fw_state & INPROGRESS) == 0)
 		WORKLIST_INSERT(&freeblks->fb_freeworkhd, &freework->fw_list);
 	if ((freeblks->fb_state &
 	    (ONWORKLIST | INPROGRESS | ALLCOMPLETE)) == ALLCOMPLETE &&
 	    LIST_EMPTY(&freeblks->fb_jblkdephd))
 		add_to_worklist(&freeblks->fb_list, WK_NODELAY);
 }
 
 /*
  * Start, continue, or finish the process of freeing an indirect block tree.
  * The free operation may be paused at any point with fw_off containing the
  * offset to restart from.  This enables us to implement some flow control
  * for large truncates which may fan out and generate a huge number of
  * dependencies.
  */
 static void
 handle_workitem_indirblk(freework)
 	struct freework *freework;
 {
 	struct freeblks *freeblks;
 	struct ufsmount *ump;
 	struct fs *fs;
 
 	freeblks = freework->fw_freeblks;
 	ump = VFSTOUFS(freeblks->fb_list.wk_mp);
 	fs = ump->um_fs;
 	if (freework->fw_state & DEPCOMPLETE) {
 		handle_written_freework(freework);
 		return;
 	}
 	if (freework->fw_off == NINDIR(fs)) {
 		freework_freeblock(freework);
 		return;
 	}
 	freework->fw_state |= INPROGRESS;
 	FREE_LOCK(ump);
 	indir_trunc(freework, fsbtodb(fs, freework->fw_blkno),
 	    freework->fw_lbn);
 	ACQUIRE_LOCK(ump);
 }
 
 /*
  * Called when a freework structure attached to a cg buf is written.  The
  * ref on either the parent or the freeblks structure is released and
  * the freeblks is added back to the worklist if there is more work to do.
  */
 static void
 handle_written_freework(freework)
 	struct freework *freework;
 {
 	struct freeblks *freeblks;
 	struct freework *parent;
 
 	freeblks = freework->fw_freeblks;
 	parent = freework->fw_parent;
 	if (freework->fw_state & DELAYEDFREE)
 		freeblks->fb_cgwait--;
 	freework->fw_state |= COMPLETE;
 	if ((freework->fw_state & ALLCOMPLETE) == ALLCOMPLETE)
 		WORKITEM_FREE(freework, D_FREEWORK);
 	if (parent) {
 		if (--parent->fw_ref == 0)
 			freework_enqueue(parent);
 		return;
 	}
 	if (--freeblks->fb_ref != 0)
 		return;
 	if ((freeblks->fb_state & (ALLCOMPLETE | ONWORKLIST | INPROGRESS)) ==
 	    ALLCOMPLETE && LIST_EMPTY(&freeblks->fb_jblkdephd)) 
 		add_to_worklist(&freeblks->fb_list, WK_NODELAY);
 }
 
 /*
  * This workitem routine performs the block de-allocation.
  * The workitem is added to the pending list after the updated
  * inode block has been written to disk.  As mentioned above,
  * checks regarding the number of blocks de-allocated (compared
  * to the number of blocks allocated for the file) are also
  * performed in this function.
  */
 static int
 handle_workitem_freeblocks(freeblks, flags)
 	struct freeblks *freeblks;
 	int flags;
 {
 	struct freework *freework;
 	struct newblk *newblk;
 	struct allocindir *aip;
 	struct ufsmount *ump;
 	struct worklist *wk;
 
 	KASSERT(LIST_EMPTY(&freeblks->fb_jblkdephd),
 	    ("handle_workitem_freeblocks: Journal entries not written."));
 	ump = VFSTOUFS(freeblks->fb_list.wk_mp);
 	ACQUIRE_LOCK(ump);
 	while ((wk = LIST_FIRST(&freeblks->fb_freeworkhd)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		switch (wk->wk_type) {
 		case D_DIRREM:
 			wk->wk_state |= COMPLETE;
 			add_to_worklist(wk, 0);
 			continue;
 
 		case D_ALLOCDIRECT:
 			free_newblk(WK_NEWBLK(wk));
 			continue;
 
 		case D_ALLOCINDIR:
 			aip = WK_ALLOCINDIR(wk);
 			freework = NULL;
 			if (aip->ai_state & DELAYEDFREE) {
 				FREE_LOCK(ump);
 				freework = newfreework(ump, freeblks, NULL,
 				    aip->ai_lbn, aip->ai_newblkno,
 				    ump->um_fs->fs_frag, 0, 0);
 				ACQUIRE_LOCK(ump);
 			}
 			newblk = WK_NEWBLK(wk);
 			if (newblk->nb_jnewblk) {
 				freework->fw_jnewblk = newblk->nb_jnewblk;
 				newblk->nb_jnewblk->jn_dep = &freework->fw_list;
 				newblk->nb_jnewblk = NULL;
 			}
 			free_newblk(newblk);
 			continue;
 
 		case D_FREEWORK:
 			freework = WK_FREEWORK(wk);
 			if (freework->fw_lbn <= -NDADDR)
 				handle_workitem_indirblk(freework);
 			else
 				freework_freeblock(freework);
 			continue;
 		default:
 			panic("handle_workitem_freeblocks: Unknown type %s",
 			    TYPENAME(wk->wk_type));
 		}
 	}
 	if (freeblks->fb_ref != 0) {
 		freeblks->fb_state &= ~INPROGRESS;
 		wake_worklist(&freeblks->fb_list);
 		freeblks = NULL;
 	}
 	FREE_LOCK(ump);
 	if (freeblks)
 		return handle_complete_freeblocks(freeblks, flags);
 	return (0);
 }
 
 /*
  * Handle completion of block free via truncate.  This allows fs_pending
  * to track the actual free block count more closely than if we only updated
  * it at the end.  We must be careful to handle cases where the block count
  * on free was incorrect.
  */
 static void
 freeblks_free(ump, freeblks, blocks)
 	struct ufsmount *ump;
 	struct freeblks *freeblks;
 	int blocks;
 {
 	struct fs *fs;
 	ufs2_daddr_t remain;
 
 	UFS_LOCK(ump);
 	remain = -freeblks->fb_chkcnt;
 	freeblks->fb_chkcnt += blocks;
 	if (remain > 0) {
 		if (remain < blocks)
 			blocks = remain;
 		fs = ump->um_fs;
 		fs->fs_pendingblocks -= blocks;
 	}
 	UFS_UNLOCK(ump);
 }
 
 /*
  * Once all of the freework workitems are complete we can retire the
  * freeblocks dependency and any journal work awaiting completion.  This
  * can not be called until all other dependencies are stable on disk.
  */
 static int
 handle_complete_freeblocks(freeblks, flags)
 	struct freeblks *freeblks;
 	int flags;
 {
 	struct inodedep *inodedep;
 	struct inode *ip;
 	struct vnode *vp;
 	struct fs *fs;
 	struct ufsmount *ump;
 	ufs2_daddr_t spare;
 
 	ump = VFSTOUFS(freeblks->fb_list.wk_mp);
 	fs = ump->um_fs;
 	flags = LK_EXCLUSIVE | flags;
 	spare = freeblks->fb_chkcnt;
 
 	/*
 	 * If we did not release the expected number of blocks we may have
 	 * to adjust the inode block count here.  Only do so if it wasn't
 	 * a truncation to zero and the modrev still matches.
 	 */
 	if (spare && freeblks->fb_len != 0) {
 		if (ffs_vgetf(freeblks->fb_list.wk_mp, freeblks->fb_inum,
 		    flags, &vp, FFSV_FORCEINSMQ) != 0)
 			return (EBUSY);
 		ip = VTOI(vp);
 		if (DIP(ip, i_modrev) == freeblks->fb_modrev) {
 			DIP_SET(ip, i_blocks, DIP(ip, i_blocks) - spare);
 			ip->i_flag |= IN_CHANGE;
 			/*
 			 * We must wait so this happens before the
 			 * journal is reclaimed.
 			 */
 			ffs_update(vp, 1);
 		}
 		vput(vp);
 	}
 	if (spare < 0) {
 		UFS_LOCK(ump);
 		fs->fs_pendingblocks += spare;
 		UFS_UNLOCK(ump);
 	}
 #ifdef QUOTA
 	/* Handle spare. */
 	if (spare)
 		quotaadj(freeblks->fb_quota, ump, -spare);
 	quotarele(freeblks->fb_quota);
 #endif
 	ACQUIRE_LOCK(ump);
 	if (freeblks->fb_state & ONDEPLIST) {
 		inodedep_lookup(freeblks->fb_list.wk_mp, freeblks->fb_inum,
 		    0, &inodedep);
 		TAILQ_REMOVE(&inodedep->id_freeblklst, freeblks, fb_next);
 		freeblks->fb_state &= ~ONDEPLIST;
 		if (TAILQ_EMPTY(&inodedep->id_freeblklst))
 			free_inodedep(inodedep);
 	}
 	/*
 	 * All of the freeblock deps must be complete prior to this call
 	 * so it's now safe to complete earlier outstanding journal entries.
 	 */
 	handle_jwork(&freeblks->fb_jwork);
 	WORKITEM_FREE(freeblks, D_FREEBLKS);
 	FREE_LOCK(ump);
 	return (0);
 }
 
 /*
  * Release blocks associated with the freeblks and stored in the indirect
  * block dbn. If level is greater than SINGLE, the block is an indirect block
  * and recursive calls to indirtrunc must be used to cleanse other indirect
  * blocks.
  *
  * This handles partial and complete truncation of blocks.  Partial is noted
  * with goingaway == 0.  In this case the freework is completed after the
  * zero'd indirects are written to disk.  For full truncation the freework
  * is completed after the block is freed.
  */
 static void
 indir_trunc(freework, dbn, lbn)
 	struct freework *freework;
 	ufs2_daddr_t dbn;
 	ufs_lbn_t lbn;
 {
 	struct freework *nfreework;
 	struct workhead wkhd;
 	struct freeblks *freeblks;
 	struct buf *bp;
 	struct fs *fs;
 	struct indirdep *indirdep;
 	struct ufsmount *ump;
 	ufs1_daddr_t *bap1 = 0;
 	ufs2_daddr_t nb, nnb, *bap2 = 0;
 	ufs_lbn_t lbnadd, nlbn;
 	int i, nblocks, ufs1fmt;
 	int freedblocks;
 	int goingaway;
 	int freedeps;
 	int needj;
 	int level;
 	int cnt;
 
 	freeblks = freework->fw_freeblks;
 	ump = VFSTOUFS(freeblks->fb_list.wk_mp);
 	fs = ump->um_fs;
 	/*
 	 * Get buffer of block pointers to be freed.  There are three cases:
 	 * 
 	 * 1) Partial truncate caches the indirdep pointer in the freework
 	 *    which provides us a back copy to the save bp which holds the
 	 *    pointers we want to clear.  When this completes the zero
 	 *    pointers are written to the real copy.
 	 * 2) The indirect is being completely truncated, cancel_indirdep()
 	 *    eliminated the real copy and placed the indirdep on the saved
 	 *    copy.  The indirdep and buf are discarded when this completes.
 	 * 3) The indirect was not in memory, we read a copy off of the disk
 	 *    using the devvp and drop and invalidate the buffer when we're
 	 *    done.
 	 */
 	goingaway = 1;
 	indirdep = NULL;
 	if (freework->fw_indir != NULL) {
 		goingaway = 0;
 		indirdep = freework->fw_indir;
 		bp = indirdep->ir_savebp;
 		if (bp == NULL || bp->b_blkno != dbn)
 			panic("indir_trunc: Bad saved buf %p blkno %jd",
 			    bp, (intmax_t)dbn);
 	} else if ((bp = incore(&freeblks->fb_devvp->v_bufobj, dbn)) != NULL) {
 		/*
 		 * The lock prevents the buf dep list from changing and
 	 	 * indirects on devvp should only ever have one dependency.
 		 */
 		indirdep = WK_INDIRDEP(LIST_FIRST(&bp->b_dep));
 		if (indirdep == NULL || (indirdep->ir_state & GOINGAWAY) == 0)
 			panic("indir_trunc: Bad indirdep %p from buf %p",
 			    indirdep, bp);
 	} else if (bread(freeblks->fb_devvp, dbn, (int)fs->fs_bsize,
 	    NOCRED, &bp) != 0) {
 		brelse(bp);
 		return;
 	}
 	ACQUIRE_LOCK(ump);
 	/* Protects against a race with complete_trunc_indir(). */
 	freework->fw_state &= ~INPROGRESS;
 	/*
 	 * If we have an indirdep we need to enforce the truncation order
 	 * and discard it when it is complete.
 	 */
 	if (indirdep) {
 		if (freework != TAILQ_FIRST(&indirdep->ir_trunc) &&
 		    !TAILQ_EMPTY(&indirdep->ir_trunc)) {
 			/*
 			 * Add the complete truncate to the list on the
 			 * indirdep to enforce in-order processing.
 			 */
 			if (freework->fw_indir == NULL)
 				TAILQ_INSERT_TAIL(&indirdep->ir_trunc,
 				    freework, fw_next);
 			FREE_LOCK(ump);
 			return;
 		}
 		/*
 		 * If we're goingaway, free the indirdep.  Otherwise it will
 		 * linger until the write completes.
 		 */
 		if (goingaway)
 			free_indirdep(indirdep);
 	}
 	FREE_LOCK(ump);
 	/* Initialize pointers depending on block size. */
 	if (ump->um_fstype == UFS1) {
 		bap1 = (ufs1_daddr_t *)bp->b_data;
 		nb = bap1[freework->fw_off];
 		ufs1fmt = 1;
 	} else {
 		bap2 = (ufs2_daddr_t *)bp->b_data;
 		nb = bap2[freework->fw_off];
 		ufs1fmt = 0;
 	}
 	level = lbn_level(lbn);
 	needj = MOUNTEDSUJ(UFSTOVFS(ump)) != 0;
 	lbnadd = lbn_offset(fs, level);
 	nblocks = btodb(fs->fs_bsize);
 	nfreework = freework;
 	freedeps = 0;
 	cnt = 0;
 	/*
 	 * Reclaim blocks.  Traverses into nested indirect levels and
 	 * arranges for the current level to be freed when subordinates
 	 * are free when journaling.
 	 */
 	for (i = freework->fw_off; i < NINDIR(fs); i++, nb = nnb) {
 		if (i != NINDIR(fs) - 1) {
 			if (ufs1fmt)
 				nnb = bap1[i+1];
 			else
 				nnb = bap2[i+1];
 		} else
 			nnb = 0;
 		if (nb == 0)
 			continue;
 		cnt++;
 		if (level != 0) {
 			nlbn = (lbn + 1) - (i * lbnadd);
 			if (needj != 0) {
 				nfreework = newfreework(ump, freeblks, freework,
 				    nlbn, nb, fs->fs_frag, 0, 0);
 				freedeps++;
 			}
 			indir_trunc(nfreework, fsbtodb(fs, nb), nlbn);
 		} else {
 			struct freedep *freedep;
 
 			/*
 			 * Attempt to aggregate freedep dependencies for
 			 * all blocks being released to the same CG.
 			 */
 			LIST_INIT(&wkhd);
 			if (needj != 0 &&
 			    (nnb == 0 || (dtog(fs, nb) != dtog(fs, nnb)))) {
 				freedep = newfreedep(freework);
 				WORKLIST_INSERT_UNLOCKED(&wkhd,
 				    &freedep->fd_list);
 				freedeps++;
 			}
 			CTR3(KTR_SUJ,
 			    "indir_trunc: ino %d blkno %jd size %ld",
 			    freeblks->fb_inum, nb, fs->fs_bsize);
 			ffs_blkfree(ump, fs, freeblks->fb_devvp, nb,
 			    fs->fs_bsize, freeblks->fb_inum,
 			    freeblks->fb_vtype, &wkhd);
 		}
 	}
 	if (goingaway) {
 		bp->b_flags |= B_INVAL | B_NOCACHE;
 		brelse(bp);
 	}
 	freedblocks = 0;
 	if (level == 0)
 		freedblocks = (nblocks * cnt);
 	if (needj == 0)
 		freedblocks += nblocks;
 	freeblks_free(ump, freeblks, freedblocks);
 	/*
 	 * If we are journaling set up the ref counts and offset so this
 	 * indirect can be completed when its children are free.
 	 */
 	if (needj) {
 		ACQUIRE_LOCK(ump);
 		freework->fw_off = i;
 		freework->fw_ref += freedeps;
 		freework->fw_ref -= NINDIR(fs) + 1;
 		if (level == 0)
 			freeblks->fb_cgwait += freedeps;
 		if (freework->fw_ref == 0)
 			freework_freeblock(freework);
 		FREE_LOCK(ump);
 		return;
 	}
 	/*
 	 * If we're not journaling we can free the indirect now.
 	 */
 	dbn = dbtofsb(fs, dbn);
 	CTR3(KTR_SUJ,
 	    "indir_trunc 2: ino %d blkno %jd size %ld",
 	    freeblks->fb_inum, dbn, fs->fs_bsize);
 	ffs_blkfree(ump, fs, freeblks->fb_devvp, dbn, fs->fs_bsize,
 	    freeblks->fb_inum, freeblks->fb_vtype, NULL);
 	/* Non SUJ softdep does single-threaded truncations. */
 	if (freework->fw_blkno == dbn) {
 		freework->fw_state |= ALLCOMPLETE;
 		ACQUIRE_LOCK(ump);
 		handle_written_freework(freework);
 		FREE_LOCK(ump);
 	}
 	return;
 }
 
 /*
  * Cancel an allocindir when it is removed via truncation.  When bp is not
  * NULL the indirect never appeared on disk and is scheduled to be freed
  * independently of the indir so we can more easily track journal work.
  */
 static void
 cancel_allocindir(aip, bp, freeblks, trunc)
 	struct allocindir *aip;
 	struct buf *bp;
 	struct freeblks *freeblks;
 	int trunc;
 {
 	struct indirdep *indirdep;
 	struct freefrag *freefrag;
 	struct newblk *newblk;
 
 	newblk = (struct newblk *)aip;
 	LIST_REMOVE(aip, ai_next);
 	/*
 	 * We must eliminate the pointer in bp if it must be freed on its
 	 * own due to partial truncate or pending journal work.
 	 */
 	if (bp && (trunc || newblk->nb_jnewblk)) {
 		/*
 		 * Clear the pointer and mark the aip to be freed
 		 * directly if it never existed on disk.
 		 */
 		aip->ai_state |= DELAYEDFREE;
 		indirdep = aip->ai_indirdep;
 		if (indirdep->ir_state & UFS1FMT)
 			((ufs1_daddr_t *)bp->b_data)[aip->ai_offset] = 0;
 		else
 			((ufs2_daddr_t *)bp->b_data)[aip->ai_offset] = 0;
 	}
 	/*
 	 * When truncating the previous pointer will be freed via
 	 * savedbp.  Eliminate the freefrag which would dup free.
 	 */
 	if (trunc && (freefrag = newblk->nb_freefrag) != NULL) {
 		newblk->nb_freefrag = NULL;
 		if (freefrag->ff_jdep)
 			cancel_jfreefrag(
 			    WK_JFREEFRAG(freefrag->ff_jdep));
 		jwork_move(&freeblks->fb_jwork, &freefrag->ff_jwork);
 		WORKITEM_FREE(freefrag, D_FREEFRAG);
 	}
 	/*
 	 * If the journal hasn't been written the jnewblk must be passed
 	 * to the call to ffs_blkfree that reclaims the space.  We accomplish
 	 * this by leaving the journal dependency on the newblk to be freed
 	 * when a freework is created in handle_workitem_freeblocks().
 	 */
 	cancel_newblk(newblk, NULL, &freeblks->fb_jwork);
 	WORKLIST_INSERT(&freeblks->fb_freeworkhd, &newblk->nb_list);
 }
 
 /*
  * Create the mkdir dependencies for . and .. in a new directory.  Link them
  * in to a newdirblk so any subsequent additions are tracked properly.  The
  * caller is responsible for adding the mkdir1 dependency to the journal
  * and updating id_mkdiradd.  This function returns with the per-filesystem
  * lock held.
  */
 static struct mkdir *
 setup_newdir(dap, newinum, dinum, newdirbp, mkdirp)
 	struct diradd *dap;
 	ino_t newinum;
 	ino_t dinum;
 	struct buf *newdirbp;
 	struct mkdir **mkdirp;
 {
 	struct newblk *newblk;
 	struct pagedep *pagedep;
 	struct inodedep *inodedep;
 	struct newdirblk *newdirblk = 0;
 	struct mkdir *mkdir1, *mkdir2;
 	struct worklist *wk;
 	struct jaddref *jaddref;
 	struct ufsmount *ump;
 	struct mount *mp;
 
 	mp = dap->da_list.wk_mp;
 	ump = VFSTOUFS(mp);
 	newdirblk = malloc(sizeof(struct newdirblk), M_NEWDIRBLK,
 	    M_SOFTDEP_FLAGS);
 	workitem_alloc(&newdirblk->db_list, D_NEWDIRBLK, mp);
 	LIST_INIT(&newdirblk->db_mkdir);
 	mkdir1 = malloc(sizeof(struct mkdir), M_MKDIR, M_SOFTDEP_FLAGS);
 	workitem_alloc(&mkdir1->md_list, D_MKDIR, mp);
 	mkdir1->md_state = ATTACHED | MKDIR_BODY;
 	mkdir1->md_diradd = dap;
 	mkdir1->md_jaddref = NULL;
 	mkdir2 = malloc(sizeof(struct mkdir), M_MKDIR, M_SOFTDEP_FLAGS);
 	workitem_alloc(&mkdir2->md_list, D_MKDIR, mp);
 	mkdir2->md_state = ATTACHED | MKDIR_PARENT;
 	mkdir2->md_diradd = dap;
 	mkdir2->md_jaddref = NULL;
 	if (MOUNTEDSUJ(mp) == 0) {
 		mkdir1->md_state |= DEPCOMPLETE;
 		mkdir2->md_state |= DEPCOMPLETE;
 	}
 	/*
 	 * Dependency on "." and ".." being written to disk.
 	 */
 	mkdir1->md_buf = newdirbp;
 	ACQUIRE_LOCK(VFSTOUFS(mp));
 	LIST_INSERT_HEAD(&ump->softdep_mkdirlisthd, mkdir1, md_mkdirs);
 	/*
 	 * We must link the pagedep, allocdirect, and newdirblk for
 	 * the initial file page so the pointer to the new directory
 	 * is not written until the directory contents are live and
 	 * any subsequent additions are not marked live until the
 	 * block is reachable via the inode.
 	 */
 	if (pagedep_lookup(mp, newdirbp, newinum, 0, 0, &pagedep) == 0)
 		panic("setup_newdir: lost pagedep");
 	LIST_FOREACH(wk, &newdirbp->b_dep, wk_list)
 		if (wk->wk_type == D_ALLOCDIRECT)
 			break;
 	if (wk == NULL)
 		panic("setup_newdir: lost allocdirect");
 	if (pagedep->pd_state & NEWBLOCK)
 		panic("setup_newdir: NEWBLOCK already set");
 	newblk = WK_NEWBLK(wk);
 	pagedep->pd_state |= NEWBLOCK;
 	pagedep->pd_newdirblk = newdirblk;
 	newdirblk->db_pagedep = pagedep;
 	WORKLIST_INSERT(&newblk->nb_newdirblk, &newdirblk->db_list);
 	WORKLIST_INSERT(&newdirblk->db_mkdir, &mkdir1->md_list);
 	/*
 	 * Look up the inodedep for the parent directory so that we
 	 * can link mkdir2 into the pending dotdot jaddref or
 	 * the inode write if there is none.  If the inode is
 	 * ALLCOMPLETE and no jaddref is present all dependencies have
 	 * been satisfied and mkdir2 can be freed.
 	 */
 	inodedep_lookup(mp, dinum, 0, &inodedep);
 	if (MOUNTEDSUJ(mp)) {
 		if (inodedep == NULL)
 			panic("setup_newdir: Lost parent.");
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref != NULL && jaddref->ja_parent == newinum &&
 		    (jaddref->ja_state & MKDIR_PARENT),
 		    ("setup_newdir: bad dotdot jaddref %p", jaddref));
 		LIST_INSERT_HEAD(&ump->softdep_mkdirlisthd, mkdir2, md_mkdirs);
 		mkdir2->md_jaddref = jaddref;
 		jaddref->ja_mkdir = mkdir2;
 	} else if (inodedep == NULL ||
 	    (inodedep->id_state & ALLCOMPLETE) == ALLCOMPLETE) {
 		dap->da_state &= ~MKDIR_PARENT;
 		WORKITEM_FREE(mkdir2, D_MKDIR);
 		mkdir2 = NULL;
 	} else {
 		LIST_INSERT_HEAD(&ump->softdep_mkdirlisthd, mkdir2, md_mkdirs);
 		WORKLIST_INSERT(&inodedep->id_bufwait, &mkdir2->md_list);
 	}
 	*mkdirp = mkdir2;
 
 	return (mkdir1);
 }
 
 /*
  * Directory entry addition dependencies.
  * 
  * When adding a new directory entry, the inode (with its incremented link
  * count) must be written to disk before the directory entry's pointer to it.
  * Also, if the inode is newly allocated, the corresponding freemap must be
  * updated (on disk) before the directory entry's pointer. These requirements
  * are met via undo/redo on the directory entry's pointer, which consists
  * simply of the inode number.
  * 
  * As directory entries are added and deleted, the free space within a
  * directory block can become fragmented.  The ufs filesystem will compact
  * a fragmented directory block to make space for a new entry. When this
  * occurs, the offsets of previously added entries change. Any "diradd"
  * dependency structures corresponding to these entries must be updated with
  * the new offsets.
  */
 
 /*
  * This routine is called after the in-memory inode's link
  * count has been incremented, but before the directory entry's
  * pointer to the inode has been set.
  */
 int
 softdep_setup_directory_add(bp, dp, diroffset, newinum, newdirbp, isnewblk)
 	struct buf *bp;		/* buffer containing directory block */
 	struct inode *dp;	/* inode for directory */
 	off_t diroffset;	/* offset of new entry in directory */
 	ino_t newinum;		/* inode referenced by new directory entry */
 	struct buf *newdirbp;	/* non-NULL => contents of new mkdir */
 	int isnewblk;		/* entry is in a newly allocated block */
 {
 	int offset;		/* offset of new entry within directory block */
 	ufs_lbn_t lbn;		/* block in directory containing new entry */
 	struct fs *fs;
 	struct diradd *dap;
 	struct newblk *newblk;
 	struct pagedep *pagedep;
 	struct inodedep *inodedep;
 	struct newdirblk *newdirblk = 0;
 	struct mkdir *mkdir1, *mkdir2;
 	struct jaddref *jaddref;
 	struct ufsmount *ump;
 	struct mount *mp;
 	int isindir;
 
 	ump = dp->i_ump;
 	mp = UFSTOVFS(ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_directory_add called on non-softdep filesystem"));
 	/*
 	 * Whiteouts have no dependencies.
 	 */
 	if (newinum == WINO) {
 		if (newdirbp != NULL)
 			bdwrite(newdirbp);
 		return (0);
 	}
 	jaddref = NULL;
 	mkdir1 = mkdir2 = NULL;
 	fs = dp->i_fs;
 	lbn = lblkno(fs, diroffset);
 	offset = blkoff(fs, diroffset);
 	dap = malloc(sizeof(struct diradd), M_DIRADD,
 		M_SOFTDEP_FLAGS|M_ZERO);
 	workitem_alloc(&dap->da_list, D_DIRADD, mp);
 	dap->da_offset = offset;
 	dap->da_newinum = newinum;
 	dap->da_state = ATTACHED;
 	LIST_INIT(&dap->da_jwork);
 	isindir = bp->b_lblkno >= NDADDR;
 	if (isnewblk &&
 	    (isindir ? blkoff(fs, diroffset) : fragoff(fs, diroffset)) == 0) {
 		newdirblk = malloc(sizeof(struct newdirblk),
 		    M_NEWDIRBLK, M_SOFTDEP_FLAGS);
 		workitem_alloc(&newdirblk->db_list, D_NEWDIRBLK, mp);
 		LIST_INIT(&newdirblk->db_mkdir);
 	}
 	/*
 	 * If we're creating a new directory setup the dependencies and set
 	 * the dap state to wait for them.  Otherwise it's COMPLETE and
 	 * we can move on.
 	 */
 	if (newdirbp == NULL) {
 		dap->da_state |= DEPCOMPLETE;
 		ACQUIRE_LOCK(ump);
 	} else {
 		dap->da_state |= MKDIR_BODY | MKDIR_PARENT;
 		mkdir1 = setup_newdir(dap, newinum, dp->i_number, newdirbp,
 		    &mkdir2);
 	}
 	/*
 	 * Link into parent directory pagedep to await its being written.
 	 */
 	pagedep_lookup(mp, bp, dp->i_number, lbn, DEPALLOC, &pagedep);
 #ifdef DEBUG
 	if (diradd_lookup(pagedep, offset) != NULL)
 		panic("softdep_setup_directory_add: %p already at off %d\n",
 		    diradd_lookup(pagedep, offset), offset);
 #endif
 	dap->da_pagedep = pagedep;
 	LIST_INSERT_HEAD(&pagedep->pd_diraddhd[DIRADDHASH(offset)], dap,
 	    da_pdlist);
 	inodedep_lookup(mp, newinum, DEPALLOC | NODELAY, &inodedep);
 	/*
 	 * If we're journaling, link the diradd into the jaddref so it
 	 * may be completed after the journal entry is written.  Otherwise,
 	 * link the diradd into its inodedep.  If the inode is not yet
 	 * written place it on the bufwait list, otherwise do the post-inode
 	 * write processing to put it on the id_pendinghd list.
 	 */
 	if (MOUNTEDSUJ(mp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref != NULL && jaddref->ja_parent == dp->i_number,
 		    ("softdep_setup_directory_add: bad jaddref %p", jaddref));
 		jaddref->ja_diroff = diroffset;
 		jaddref->ja_diradd = dap;
 		add_to_journal(&jaddref->ja_list);
 	} else if ((inodedep->id_state & ALLCOMPLETE) == ALLCOMPLETE)
 		diradd_inode_written(dap, inodedep);
 	else
 		WORKLIST_INSERT(&inodedep->id_bufwait, &dap->da_list);
 	/*
 	 * Add the journal entries for . and .. links now that the primary
 	 * link is written.
 	 */
 	if (mkdir1 != NULL && MOUNTEDSUJ(mp)) {
 		jaddref = (struct jaddref *)TAILQ_PREV(&jaddref->ja_ref,
 		    inoreflst, if_deps);
 		KASSERT(jaddref != NULL &&
 		    jaddref->ja_ino == jaddref->ja_parent &&
 		    (jaddref->ja_state & MKDIR_BODY),
 		    ("softdep_setup_directory_add: bad dot jaddref %p",
 		    jaddref));
 		mkdir1->md_jaddref = jaddref;
 		jaddref->ja_mkdir = mkdir1;
 		/*
 		 * It is important that the dotdot journal entry
 		 * is added prior to the dot entry since dot writes
 		 * both the dot and dotdot links.  These both must
 		 * be added after the primary link for the journal
 		 * to remain consistent.
 		 */
 		add_to_journal(&mkdir2->md_jaddref->ja_list);
 		add_to_journal(&jaddref->ja_list);
 	}
 	/*
 	 * If we are adding a new directory remember this diradd so that if
 	 * we rename it we can keep the dot and dotdot dependencies.  If
 	 * we are adding a new name for an inode that has a mkdiradd we
 	 * must be in rename and we have to move the dot and dotdot
 	 * dependencies to this new name.  The old name is being orphaned
 	 * soon.
 	 */
 	if (mkdir1 != NULL) {
 		if (inodedep->id_mkdiradd != NULL)
 			panic("softdep_setup_directory_add: Existing mkdir");
 		inodedep->id_mkdiradd = dap;
 	} else if (inodedep->id_mkdiradd)
 		merge_diradd(inodedep, dap);
 	if (newdirblk) {
 		/*
 		 * There is nothing to do if we are already tracking
 		 * this block.
 		 */
 		if ((pagedep->pd_state & NEWBLOCK) != 0) {
 			WORKITEM_FREE(newdirblk, D_NEWDIRBLK);
 			FREE_LOCK(ump);
 			return (0);
 		}
 		if (newblk_lookup(mp, dbtofsb(fs, bp->b_blkno), 0, &newblk)
 		    == 0)
 			panic("softdep_setup_directory_add: lost entry");
 		WORKLIST_INSERT(&newblk->nb_newdirblk, &newdirblk->db_list);
 		pagedep->pd_state |= NEWBLOCK;
 		pagedep->pd_newdirblk = newdirblk;
 		newdirblk->db_pagedep = pagedep;
 		FREE_LOCK(ump);
 		/*
 		 * If we extended into an indirect signal direnter to sync.
 		 */
 		if (isindir)
 			return (1);
 		return (0);
 	}
 	FREE_LOCK(ump);
 	return (0);
 }
 
 /*
  * This procedure is called to change the offset of a directory
  * entry when compacting a directory block which must be owned
  * exclusively by the caller. Note that the actual entry movement
  * must be done in this procedure to ensure that no I/O completions
  * occur while the move is in progress.
  */
 void 
 softdep_change_directoryentry_offset(bp, dp, base, oldloc, newloc, entrysize)
 	struct buf *bp;		/* Buffer holding directory block. */
 	struct inode *dp;	/* inode for directory */
 	caddr_t base;		/* address of dp->i_offset */
 	caddr_t oldloc;		/* address of old directory location */
 	caddr_t newloc;		/* address of new directory location */
 	int entrysize;		/* size of directory entry */
 {
 	int offset, oldoffset, newoffset;
 	struct pagedep *pagedep;
 	struct jmvref *jmvref;
 	struct diradd *dap;
 	struct direct *de;
 	struct mount *mp;
 	ufs_lbn_t lbn;
 	int flags;
 
 	mp = UFSTOVFS(dp->i_ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_change_directoryentry_offset called on "
 	     "non-softdep filesystem"));
 	de = (struct direct *)oldloc;
 	jmvref = NULL;
 	flags = 0;
 	/*
 	 * Moves are always journaled as it would be too complex to
 	 * determine if any affected adds or removes are present in the
 	 * journal.
 	 */
 	if (MOUNTEDSUJ(mp)) {
 		flags = DEPALLOC;
 		jmvref = newjmvref(dp, de->d_ino,
 		    dp->i_offset + (oldloc - base),
 		    dp->i_offset + (newloc - base));
 	}
 	lbn = lblkno(dp->i_fs, dp->i_offset);
 	offset = blkoff(dp->i_fs, dp->i_offset);
 	oldoffset = offset + (oldloc - base);
 	newoffset = offset + (newloc - base);
 	ACQUIRE_LOCK(dp->i_ump);
 	if (pagedep_lookup(mp, bp, dp->i_number, lbn, flags, &pagedep) == 0)
 		goto done;
 	dap = diradd_lookup(pagedep, oldoffset);
 	if (dap) {
 		dap->da_offset = newoffset;
 		newoffset = DIRADDHASH(newoffset);
 		oldoffset = DIRADDHASH(oldoffset);
 		if ((dap->da_state & ALLCOMPLETE) != ALLCOMPLETE &&
 		    newoffset != oldoffset) {
 			LIST_REMOVE(dap, da_pdlist);
 			LIST_INSERT_HEAD(&pagedep->pd_diraddhd[newoffset],
 			    dap, da_pdlist);
 		}
 	}
 done:
 	if (jmvref) {
 		jmvref->jm_pagedep = pagedep;
 		LIST_INSERT_HEAD(&pagedep->pd_jmvrefhd, jmvref, jm_deps);
 		add_to_journal(&jmvref->jm_list);
 	}
 	bcopy(oldloc, newloc, entrysize);
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Move the mkdir dependencies and journal work from one diradd to another
  * when renaming a directory.  The new name must depend on the mkdir deps
  * completing as the old name did.  Directories can only have one valid link
  * at a time so one must be canonical.
  */
 static void
 merge_diradd(inodedep, newdap)
 	struct inodedep *inodedep;
 	struct diradd *newdap;
 {
 	struct diradd *olddap;
 	struct mkdir *mkdir, *nextmd;
 	struct ufsmount *ump;
 	short state;
 
 	olddap = inodedep->id_mkdiradd;
 	inodedep->id_mkdiradd = newdap;
 	if ((olddap->da_state & (MKDIR_PARENT | MKDIR_BODY)) != 0) {
 		newdap->da_state &= ~DEPCOMPLETE;
 		ump = VFSTOUFS(inodedep->id_list.wk_mp);
 		for (mkdir = LIST_FIRST(&ump->softdep_mkdirlisthd); mkdir;
 		     mkdir = nextmd) {
 			nextmd = LIST_NEXT(mkdir, md_mkdirs);
 			if (mkdir->md_diradd != olddap)
 				continue;
 			mkdir->md_diradd = newdap;
 			state = mkdir->md_state & (MKDIR_PARENT | MKDIR_BODY);
 			newdap->da_state |= state;
 			olddap->da_state &= ~state;
 			if ((olddap->da_state &
 			    (MKDIR_PARENT | MKDIR_BODY)) == 0)
 				break;
 		}
 		if ((olddap->da_state & (MKDIR_PARENT | MKDIR_BODY)) != 0)
 			panic("merge_diradd: unfound ref");
 	}
 	/*
 	 * Any mkdir related journal items are not safe to be freed until
 	 * the new name is stable.
 	 */
 	jwork_move(&newdap->da_jwork, &olddap->da_jwork);
 	olddap->da_state |= DEPCOMPLETE;
 	complete_diradd(olddap);
 }
 
 /*
  * Move the diradd to the pending list when all diradd dependencies are
  * complete.
  */
 static void
 complete_diradd(dap)
 	struct diradd *dap;
 {
 	struct pagedep *pagedep;
 
 	if ((dap->da_state & ALLCOMPLETE) == ALLCOMPLETE) {
 		if (dap->da_state & DIRCHG)
 			pagedep = dap->da_previous->dm_pagedep;
 		else
 			pagedep = dap->da_pagedep;
 		LIST_REMOVE(dap, da_pdlist);
 		LIST_INSERT_HEAD(&pagedep->pd_pendinghd, dap, da_pdlist);
 	}
 }
 
 /*
  * Cancel a diradd when a dirrem overlaps with it.  We must cancel the journal
  * add entries and conditonally journal the remove.
  */
 static void
 cancel_diradd(dap, dirrem, jremref, dotremref, dotdotremref)
 	struct diradd *dap;
 	struct dirrem *dirrem;
 	struct jremref *jremref;
 	struct jremref *dotremref;
 	struct jremref *dotdotremref;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	struct mkdir *mkdir;
 
 	/*
 	 * If no remove references were allocated we're on a non-journaled
 	 * filesystem and can skip the cancel step.
 	 */
 	if (jremref == NULL) {
 		free_diradd(dap, NULL);
 		return;
 	}
 	/*
 	 * Cancel the primary name an free it if it does not require
 	 * journaling.
 	 */
 	if (inodedep_lookup(dap->da_list.wk_mp, dap->da_newinum,
 	    0, &inodedep) != 0) {
 		/* Abort the addref that reference this diradd.  */
 		TAILQ_FOREACH(inoref, &inodedep->id_inoreflst, if_deps) {
 			if (inoref->if_list.wk_type != D_JADDREF)
 				continue;
 			jaddref = (struct jaddref *)inoref;
 			if (jaddref->ja_diradd != dap)
 				continue;
 			if (cancel_jaddref(jaddref, inodedep,
 			    &dirrem->dm_jwork) == 0) {
 				free_jremref(jremref);
 				jremref = NULL;
 			}
 			break;
 		}
 	}
 	/*
 	 * Cancel subordinate names and free them if they do not require
 	 * journaling.
 	 */
 	if ((dap->da_state & (MKDIR_PARENT | MKDIR_BODY)) != 0) {
 		ump = VFSTOUFS(dap->da_list.wk_mp);
 		LIST_FOREACH(mkdir, &ump->softdep_mkdirlisthd, md_mkdirs) {
 			if (mkdir->md_diradd != dap)
 				continue;
 			if ((jaddref = mkdir->md_jaddref) == NULL)
 				continue;
 			mkdir->md_jaddref = NULL;
 			if (mkdir->md_state & MKDIR_PARENT) {
 				if (cancel_jaddref(jaddref, NULL,
 				    &dirrem->dm_jwork) == 0) {
 					free_jremref(dotdotremref);
 					dotdotremref = NULL;
 				}
 			} else {
 				if (cancel_jaddref(jaddref, inodedep,
 				    &dirrem->dm_jwork) == 0) {
 					free_jremref(dotremref);
 					dotremref = NULL;
 				}
 			}
 		}
 	}
 
 	if (jremref)
 		journal_jremref(dirrem, jremref, inodedep);
 	if (dotremref)
 		journal_jremref(dirrem, dotremref, inodedep);
 	if (dotdotremref)
 		journal_jremref(dirrem, dotdotremref, NULL);
 	jwork_move(&dirrem->dm_jwork, &dap->da_jwork);
 	free_diradd(dap, &dirrem->dm_jwork);
 }
 
 /*
  * Free a diradd dependency structure. This routine must be called
  * with splbio interrupts blocked.
  */
 static void
 free_diradd(dap, wkhd)
 	struct diradd *dap;
 	struct workhead *wkhd;
 {
 	struct dirrem *dirrem;
 	struct pagedep *pagedep;
 	struct inodedep *inodedep;
 	struct mkdir *mkdir, *nextmd;
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(dap->da_list.wk_mp);
 	LOCK_OWNED(ump);
 	LIST_REMOVE(dap, da_pdlist);
 	if (dap->da_state & ONWORKLIST)
 		WORKLIST_REMOVE(&dap->da_list);
 	if ((dap->da_state & DIRCHG) == 0) {
 		pagedep = dap->da_pagedep;
 	} else {
 		dirrem = dap->da_previous;
 		pagedep = dirrem->dm_pagedep;
 		dirrem->dm_dirinum = pagedep->pd_ino;
 		dirrem->dm_state |= COMPLETE;
 		if (LIST_EMPTY(&dirrem->dm_jremrefhd))
 			add_to_worklist(&dirrem->dm_list, 0);
 	}
 	if (inodedep_lookup(pagedep->pd_list.wk_mp, dap->da_newinum,
 	    0, &inodedep) != 0)
 		if (inodedep->id_mkdiradd == dap)
 			inodedep->id_mkdiradd = NULL;
 	if ((dap->da_state & (MKDIR_PARENT | MKDIR_BODY)) != 0) {
 		for (mkdir = LIST_FIRST(&ump->softdep_mkdirlisthd); mkdir;
 		     mkdir = nextmd) {
 			nextmd = LIST_NEXT(mkdir, md_mkdirs);
 			if (mkdir->md_diradd != dap)
 				continue;
 			dap->da_state &=
 			    ~(mkdir->md_state & (MKDIR_PARENT | MKDIR_BODY));
 			LIST_REMOVE(mkdir, md_mkdirs);
 			if (mkdir->md_state & ONWORKLIST)
 				WORKLIST_REMOVE(&mkdir->md_list);
 			if (mkdir->md_jaddref != NULL)
 				panic("free_diradd: Unexpected jaddref");
 			WORKITEM_FREE(mkdir, D_MKDIR);
 			if ((dap->da_state & (MKDIR_PARENT | MKDIR_BODY)) == 0)
 				break;
 		}
 		if ((dap->da_state & (MKDIR_PARENT | MKDIR_BODY)) != 0)
 			panic("free_diradd: unfound ref");
 	}
 	if (inodedep)
 		free_inodedep(inodedep);
 	/*
 	 * Free any journal segments waiting for the directory write.
 	 */
 	handle_jwork(&dap->da_jwork);
 	WORKITEM_FREE(dap, D_DIRADD);
 }
 
 /*
  * Directory entry removal dependencies.
  * 
  * When removing a directory entry, the entry's inode pointer must be
  * zero'ed on disk before the corresponding inode's link count is decremented
  * (possibly freeing the inode for re-use). This dependency is handled by
  * updating the directory entry but delaying the inode count reduction until
  * after the directory block has been written to disk. After this point, the
  * inode count can be decremented whenever it is convenient.
  */
 
 /*
  * This routine should be called immediately after removing
  * a directory entry.  The inode's link count should not be
  * decremented by the calling procedure -- the soft updates
  * code will do this task when it is safe.
  */
 void 
 softdep_setup_remove(bp, dp, ip, isrmdir)
 	struct buf *bp;		/* buffer containing directory block */
 	struct inode *dp;	/* inode for the directory being modified */
 	struct inode *ip;	/* inode for directory entry being removed */
 	int isrmdir;		/* indicates if doing RMDIR */
 {
 	struct dirrem *dirrem, *prevdirrem;
 	struct inodedep *inodedep;
 	int direct;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_setup_remove called on non-softdep filesystem"));
 	/*
 	 * Allocate a new dirrem if appropriate and ACQUIRE_LOCK.  We want
 	 * newdirrem() to setup the full directory remove which requires
 	 * isrmdir > 1.
 	 */
 	dirrem = newdirrem(bp, dp, ip, isrmdir, &prevdirrem);
 	/*
 	 * Add the dirrem to the inodedep's pending remove list for quick
 	 * discovery later.
 	 */
 	if (inodedep_lookup(UFSTOVFS(ip->i_ump), ip->i_number, 0,
 	    &inodedep) == 0)
 		panic("softdep_setup_remove: Lost inodedep.");
 	KASSERT((inodedep->id_state & UNLINKED) == 0, ("inode unlinked"));
 	dirrem->dm_state |= ONDEPLIST;
 	LIST_INSERT_HEAD(&inodedep->id_dirremhd, dirrem, dm_inonext);
 
 	/*
 	 * If the COMPLETE flag is clear, then there were no active
 	 * entries and we want to roll back to a zeroed entry until
 	 * the new inode is committed to disk. If the COMPLETE flag is
 	 * set then we have deleted an entry that never made it to
 	 * disk. If the entry we deleted resulted from a name change,
 	 * then the old name still resides on disk. We cannot delete
 	 * its inode (returned to us in prevdirrem) until the zeroed
 	 * directory entry gets to disk. The new inode has never been
 	 * referenced on the disk, so can be deleted immediately.
 	 */
 	if ((dirrem->dm_state & COMPLETE) == 0) {
 		LIST_INSERT_HEAD(&dirrem->dm_pagedep->pd_dirremhd, dirrem,
 		    dm_next);
 		FREE_LOCK(ip->i_ump);
 	} else {
 		if (prevdirrem != NULL)
 			LIST_INSERT_HEAD(&dirrem->dm_pagedep->pd_dirremhd,
 			    prevdirrem, dm_next);
 		dirrem->dm_dirinum = dirrem->dm_pagedep->pd_ino;
 		direct = LIST_EMPTY(&dirrem->dm_jremrefhd);
 		FREE_LOCK(ip->i_ump);
 		if (direct)
 			handle_workitem_remove(dirrem, 0);
 	}
 }
 
 /*
  * Check for an entry matching 'offset' on both the pd_dirraddhd list and the
  * pd_pendinghd list of a pagedep.
  */
 static struct diradd *
 diradd_lookup(pagedep, offset)
 	struct pagedep *pagedep;
 	int offset;
 {
 	struct diradd *dap;
 
 	LIST_FOREACH(dap, &pagedep->pd_diraddhd[DIRADDHASH(offset)], da_pdlist)
 		if (dap->da_offset == offset)
 			return (dap);
 	LIST_FOREACH(dap, &pagedep->pd_pendinghd, da_pdlist)
 		if (dap->da_offset == offset)
 			return (dap);
 	return (NULL);
 }
 
 /*
  * Search for a .. diradd dependency in a directory that is being removed.
  * If the directory was renamed to a new parent we have a diradd rather
  * than a mkdir for the .. entry.  We need to cancel it now before
  * it is found in truncate().
  */
 static struct jremref *
 cancel_diradd_dotdot(ip, dirrem, jremref)
 	struct inode *ip;
 	struct dirrem *dirrem;
 	struct jremref *jremref;
 {
 	struct pagedep *pagedep;
 	struct diradd *dap;
 	struct worklist *wk;
 
 	if (pagedep_lookup(UFSTOVFS(ip->i_ump), NULL, ip->i_number, 0, 0,
 	    &pagedep) == 0)
 		return (jremref);
 	dap = diradd_lookup(pagedep, DOTDOT_OFFSET);
 	if (dap == NULL)
 		return (jremref);
 	cancel_diradd(dap, dirrem, jremref, NULL, NULL);
 	/*
 	 * Mark any journal work as belonging to the parent so it is freed
 	 * with the .. reference.
 	 */
 	LIST_FOREACH(wk, &dirrem->dm_jwork, wk_list)
 		wk->wk_state |= MKDIR_PARENT;
 	return (NULL);
 }
 
 /*
  * Cancel the MKDIR_PARENT mkdir component of a diradd when we're going to
  * replace it with a dirrem/diradd pair as a result of re-parenting a
  * directory.  This ensures that we don't simultaneously have a mkdir and
  * a diradd for the same .. entry.
  */
 static struct jremref *
 cancel_mkdir_dotdot(ip, dirrem, jremref)
 	struct inode *ip;
 	struct dirrem *dirrem;
 	struct jremref *jremref;
 {
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct ufsmount *ump;
 	struct mkdir *mkdir;
 	struct diradd *dap;
 
 	if (inodedep_lookup(UFSTOVFS(ip->i_ump), ip->i_number, 0,
 	    &inodedep) == 0)
 		return (jremref);
 	dap = inodedep->id_mkdiradd;
 	if (dap == NULL || (dap->da_state & MKDIR_PARENT) == 0)
 		return (jremref);
 	ump = VFSTOUFS(inodedep->id_list.wk_mp);
 	for (mkdir = LIST_FIRST(&ump->softdep_mkdirlisthd); mkdir;
 	    mkdir = LIST_NEXT(mkdir, md_mkdirs))
 		if (mkdir->md_diradd == dap && mkdir->md_state & MKDIR_PARENT)
 			break;
 	if (mkdir == NULL)
 		panic("cancel_mkdir_dotdot: Unable to find mkdir\n");
 	if ((jaddref = mkdir->md_jaddref) != NULL) {
 		mkdir->md_jaddref = NULL;
 		jaddref->ja_state &= ~MKDIR_PARENT;
 		if (inodedep_lookup(UFSTOVFS(ip->i_ump), jaddref->ja_ino, 0,
 		    &inodedep) == 0)
 			panic("cancel_mkdir_dotdot: Lost parent inodedep");
 		if (cancel_jaddref(jaddref, inodedep, &dirrem->dm_jwork)) {
 			journal_jremref(dirrem, jremref, inodedep);
 			jremref = NULL;
 		}
 	}
 	if (mkdir->md_state & ONWORKLIST)
 		WORKLIST_REMOVE(&mkdir->md_list);
 	mkdir->md_state |= ALLCOMPLETE;
 	complete_mkdir(mkdir);
 	return (jremref);
 }
 
 static void
 journal_jremref(dirrem, jremref, inodedep)
 	struct dirrem *dirrem;
 	struct jremref *jremref;
 	struct inodedep *inodedep;
 {
 
 	if (inodedep == NULL)
 		if (inodedep_lookup(jremref->jr_list.wk_mp,
 		    jremref->jr_ref.if_ino, 0, &inodedep) == 0)
 			panic("journal_jremref: Lost inodedep");
 	LIST_INSERT_HEAD(&dirrem->dm_jremrefhd, jremref, jr_deps);
 	TAILQ_INSERT_TAIL(&inodedep->id_inoreflst, &jremref->jr_ref, if_deps);
 	add_to_journal(&jremref->jr_list);
 }
 
 static void
 dirrem_journal(dirrem, jremref, dotremref, dotdotremref)
 	struct dirrem *dirrem;
 	struct jremref *jremref;
 	struct jremref *dotremref;
 	struct jremref *dotdotremref;
 {
 	struct inodedep *inodedep;
 
 
 	if (inodedep_lookup(jremref->jr_list.wk_mp, jremref->jr_ref.if_ino, 0,
 	    &inodedep) == 0)
 		panic("dirrem_journal: Lost inodedep");
 	journal_jremref(dirrem, jremref, inodedep);
 	if (dotremref)
 		journal_jremref(dirrem, dotremref, inodedep);
 	if (dotdotremref)
 		journal_jremref(dirrem, dotdotremref, NULL);
 }
 
 /*
  * Allocate a new dirrem if appropriate and return it along with
  * its associated pagedep. Called without a lock, returns with lock.
  */
 static struct dirrem *
 newdirrem(bp, dp, ip, isrmdir, prevdirremp)
 	struct buf *bp;		/* buffer containing directory block */
 	struct inode *dp;	/* inode for the directory being modified */
 	struct inode *ip;	/* inode for directory entry being removed */
 	int isrmdir;		/* indicates if doing RMDIR */
 	struct dirrem **prevdirremp; /* previously referenced inode, if any */
 {
 	int offset;
 	ufs_lbn_t lbn;
 	struct diradd *dap;
 	struct dirrem *dirrem;
 	struct pagedep *pagedep;
 	struct jremref *jremref;
 	struct jremref *dotremref;
 	struct jremref *dotdotremref;
 	struct vnode *dvp;
 
 	/*
 	 * Whiteouts have no deletion dependencies.
 	 */
 	if (ip == NULL)
 		panic("newdirrem: whiteout");
 	dvp = ITOV(dp);
 	/*
 	 * If the system is over its limit and our filesystem is
 	 * responsible for more than our share of that usage and
 	 * we are not a snapshot, request some inodedep cleanup.
 	 * Limiting the number of dirrem structures will also limit
 	 * the number of freefile and freeblks structures.
 	 */
 	ACQUIRE_LOCK(ip->i_ump);
 	while (!IS_SNAPSHOT(ip) && dep_current[D_DIRREM] > max_softdeps / 2 &&
 	    ip->i_ump->softdep_curdeps[D_DIRREM] >
 	    (max_softdeps / 2) / stat_flush_threads)
 		(void) request_cleanup(ITOV(dp)->v_mount, FLUSH_BLOCKS);
 	FREE_LOCK(ip->i_ump);
 	dirrem = malloc(sizeof(struct dirrem),
 		M_DIRREM, M_SOFTDEP_FLAGS|M_ZERO);
 	workitem_alloc(&dirrem->dm_list, D_DIRREM, dvp->v_mount);
 	LIST_INIT(&dirrem->dm_jremrefhd);
 	LIST_INIT(&dirrem->dm_jwork);
 	dirrem->dm_state = isrmdir ? RMDIR : 0;
 	dirrem->dm_oldinum = ip->i_number;
 	*prevdirremp = NULL;
 	/*
 	 * Allocate remove reference structures to track journal write
 	 * dependencies.  We will always have one for the link and
 	 * when doing directories we will always have one more for dot.
 	 * When renaming a directory we skip the dotdot link change so
 	 * this is not needed.
 	 */
 	jremref = dotremref = dotdotremref = NULL;
 	if (DOINGSUJ(dvp)) {
 		if (isrmdir) {
 			jremref = newjremref(dirrem, dp, ip, dp->i_offset,
 			    ip->i_effnlink + 2);
 			dotremref = newjremref(dirrem, ip, ip, DOT_OFFSET,
 			    ip->i_effnlink + 1);
 			dotdotremref = newjremref(dirrem, ip, dp, DOTDOT_OFFSET,
 			    dp->i_effnlink + 1);
 			dotdotremref->jr_state |= MKDIR_PARENT;
 		} else
 			jremref = newjremref(dirrem, dp, ip, dp->i_offset,
 			    ip->i_effnlink + 1);
 	}
 	ACQUIRE_LOCK(ip->i_ump);
 	lbn = lblkno(dp->i_fs, dp->i_offset);
 	offset = blkoff(dp->i_fs, dp->i_offset);
 	pagedep_lookup(UFSTOVFS(dp->i_ump), bp, dp->i_number, lbn, DEPALLOC,
 	    &pagedep);
 	dirrem->dm_pagedep = pagedep;
 	dirrem->dm_offset = offset;
 	/*
 	 * If we're renaming a .. link to a new directory, cancel any
 	 * existing MKDIR_PARENT mkdir.  If it has already been canceled
 	 * the jremref is preserved for any potential diradd in this
 	 * location.  This can not coincide with a rmdir.
 	 */
 	if (dp->i_offset == DOTDOT_OFFSET) {
 		if (isrmdir)
 			panic("newdirrem: .. directory change during remove?");
 		jremref = cancel_mkdir_dotdot(dp, dirrem, jremref);
 	}
 	/*
 	 * If we're removing a directory search for the .. dependency now and
 	 * cancel it.  Any pending journal work will be added to the dirrem
 	 * to be completed when the workitem remove completes.
 	 */
 	if (isrmdir)
 		dotdotremref = cancel_diradd_dotdot(ip, dirrem, dotdotremref);
 	/*
 	 * Check for a diradd dependency for the same directory entry.
 	 * If present, then both dependencies become obsolete and can
 	 * be de-allocated.
 	 */
 	dap = diradd_lookup(pagedep, offset);
 	if (dap == NULL) {
 		/*
 		 * Link the jremref structures into the dirrem so they are
 		 * written prior to the pagedep.
 		 */
 		if (jremref)
 			dirrem_journal(dirrem, jremref, dotremref,
 			    dotdotremref);
 		return (dirrem);
 	}
 	/*
 	 * Must be ATTACHED at this point.
 	 */
 	if ((dap->da_state & ATTACHED) == 0)
 		panic("newdirrem: not ATTACHED");
 	if (dap->da_newinum != ip->i_number)
 		panic("newdirrem: inum %ju should be %ju",
 		    (uintmax_t)ip->i_number, (uintmax_t)dap->da_newinum);
 	/*
 	 * If we are deleting a changed name that never made it to disk,
 	 * then return the dirrem describing the previous inode (which
 	 * represents the inode currently referenced from this entry on disk).
 	 */
 	if ((dap->da_state & DIRCHG) != 0) {
 		*prevdirremp = dap->da_previous;
 		dap->da_state &= ~DIRCHG;
 		dap->da_pagedep = pagedep;
 	}
 	/*
 	 * We are deleting an entry that never made it to disk.
 	 * Mark it COMPLETE so we can delete its inode immediately.
 	 */
 	dirrem->dm_state |= COMPLETE;
 	cancel_diradd(dap, dirrem, jremref, dotremref, dotdotremref);
 #ifdef SUJ_DEBUG
 	if (isrmdir == 0) {
 		struct worklist *wk;
 
 		LIST_FOREACH(wk, &dirrem->dm_jwork, wk_list)
 			if (wk->wk_state & (MKDIR_BODY | MKDIR_PARENT))
 				panic("bad wk %p (0x%X)\n", wk, wk->wk_state);
 	}
 #endif
 
 	return (dirrem);
 }
 
 /*
  * Directory entry change dependencies.
  * 
  * Changing an existing directory entry requires that an add operation
  * be completed first followed by a deletion. The semantics for the addition
  * are identical to the description of adding a new entry above except
  * that the rollback is to the old inode number rather than zero. Once
  * the addition dependency is completed, the removal is done as described
  * in the removal routine above.
  */
 
 /*
  * This routine should be called immediately after changing
  * a directory entry.  The inode's link count should not be
  * decremented by the calling procedure -- the soft updates
  * code will perform this task when it is safe.
  */
 void 
 softdep_setup_directory_change(bp, dp, ip, newinum, isrmdir)
 	struct buf *bp;		/* buffer containing directory block */
 	struct inode *dp;	/* inode for the directory being modified */
 	struct inode *ip;	/* inode for directory entry being removed */
 	ino_t newinum;		/* new inode number for changed entry */
 	int isrmdir;		/* indicates if doing RMDIR */
 {
 	int offset;
 	struct diradd *dap = NULL;
 	struct dirrem *dirrem, *prevdirrem;
 	struct pagedep *pagedep;
 	struct inodedep *inodedep;
 	struct jaddref *jaddref;
 	struct mount *mp;
 
 	offset = blkoff(dp->i_fs, dp->i_offset);
 	mp = UFSTOVFS(dp->i_ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	   ("softdep_setup_directory_change called on non-softdep filesystem"));
 
 	/*
 	 * Whiteouts do not need diradd dependencies.
 	 */
 	if (newinum != WINO) {
 		dap = malloc(sizeof(struct diradd),
 		    M_DIRADD, M_SOFTDEP_FLAGS|M_ZERO);
 		workitem_alloc(&dap->da_list, D_DIRADD, mp);
 		dap->da_state = DIRCHG | ATTACHED | DEPCOMPLETE;
 		dap->da_offset = offset;
 		dap->da_newinum = newinum;
 		LIST_INIT(&dap->da_jwork);
 	}
 
 	/*
 	 * Allocate a new dirrem and ACQUIRE_LOCK.
 	 */
 	dirrem = newdirrem(bp, dp, ip, isrmdir, &prevdirrem);
 	pagedep = dirrem->dm_pagedep;
 	/*
 	 * The possible values for isrmdir:
 	 *	0 - non-directory file rename
 	 *	1 - directory rename within same directory
 	 *   inum - directory rename to new directory of given inode number
 	 * When renaming to a new directory, we are both deleting and
 	 * creating a new directory entry, so the link count on the new
 	 * directory should not change. Thus we do not need the followup
 	 * dirrem which is usually done in handle_workitem_remove. We set
 	 * the DIRCHG flag to tell handle_workitem_remove to skip the 
 	 * followup dirrem.
 	 */
 	if (isrmdir > 1)
 		dirrem->dm_state |= DIRCHG;
 
 	/*
 	 * Whiteouts have no additional dependencies,
 	 * so just put the dirrem on the correct list.
 	 */
 	if (newinum == WINO) {
 		if ((dirrem->dm_state & COMPLETE) == 0) {
 			LIST_INSERT_HEAD(&pagedep->pd_dirremhd, dirrem,
 			    dm_next);
 		} else {
 			dirrem->dm_dirinum = pagedep->pd_ino;
 			if (LIST_EMPTY(&dirrem->dm_jremrefhd))
 				add_to_worklist(&dirrem->dm_list, 0);
 		}
 		FREE_LOCK(dp->i_ump);
 		return;
 	}
 	/*
 	 * Add the dirrem to the inodedep's pending remove list for quick
 	 * discovery later.  A valid nlinkdelta ensures that this lookup
 	 * will not fail.
 	 */
 	if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) == 0)
 		panic("softdep_setup_directory_change: Lost inodedep.");
 	dirrem->dm_state |= ONDEPLIST;
 	LIST_INSERT_HEAD(&inodedep->id_dirremhd, dirrem, dm_inonext);
 
 	/*
 	 * If the COMPLETE flag is clear, then there were no active
 	 * entries and we want to roll back to the previous inode until
 	 * the new inode is committed to disk. If the COMPLETE flag is
 	 * set, then we have deleted an entry that never made it to disk.
 	 * If the entry we deleted resulted from a name change, then the old
 	 * inode reference still resides on disk. Any rollback that we do
 	 * needs to be to that old inode (returned to us in prevdirrem). If
 	 * the entry we deleted resulted from a create, then there is
 	 * no entry on the disk, so we want to roll back to zero rather
 	 * than the uncommitted inode. In either of the COMPLETE cases we
 	 * want to immediately free the unwritten and unreferenced inode.
 	 */
 	if ((dirrem->dm_state & COMPLETE) == 0) {
 		dap->da_previous = dirrem;
 	} else {
 		if (prevdirrem != NULL) {
 			dap->da_previous = prevdirrem;
 		} else {
 			dap->da_state &= ~DIRCHG;
 			dap->da_pagedep = pagedep;
 		}
 		dirrem->dm_dirinum = pagedep->pd_ino;
 		if (LIST_EMPTY(&dirrem->dm_jremrefhd))
 			add_to_worklist(&dirrem->dm_list, 0);
 	}
 	/*
 	 * Lookup the jaddref for this journal entry.  We must finish
 	 * initializing it and make the diradd write dependent on it.
 	 * If we're not journaling, put it on the id_bufwait list if the
 	 * inode is not yet written. If it is written, do the post-inode
 	 * write processing to put it on the id_pendinghd list.
 	 */
 	inodedep_lookup(mp, newinum, DEPALLOC | NODELAY, &inodedep);
 	if (MOUNTEDSUJ(mp)) {
 		jaddref = (struct jaddref *)TAILQ_LAST(&inodedep->id_inoreflst,
 		    inoreflst);
 		KASSERT(jaddref != NULL && jaddref->ja_parent == dp->i_number,
 		    ("softdep_setup_directory_change: bad jaddref %p",
 		    jaddref));
 		jaddref->ja_diroff = dp->i_offset;
 		jaddref->ja_diradd = dap;
 		LIST_INSERT_HEAD(&pagedep->pd_diraddhd[DIRADDHASH(offset)],
 		    dap, da_pdlist);
 		add_to_journal(&jaddref->ja_list);
 	} else if ((inodedep->id_state & ALLCOMPLETE) == ALLCOMPLETE) {
 		dap->da_state |= COMPLETE;
 		LIST_INSERT_HEAD(&pagedep->pd_pendinghd, dap, da_pdlist);
 		WORKLIST_INSERT(&inodedep->id_pendinghd, &dap->da_list);
 	} else {
 		LIST_INSERT_HEAD(&pagedep->pd_diraddhd[DIRADDHASH(offset)],
 		    dap, da_pdlist);
 		WORKLIST_INSERT(&inodedep->id_bufwait, &dap->da_list);
 	}
 	/*
 	 * If we're making a new name for a directory that has not been
 	 * committed when need to move the dot and dotdot references to
 	 * this new name.
 	 */
 	if (inodedep->id_mkdiradd && dp->i_offset != DOTDOT_OFFSET)
 		merge_diradd(inodedep, dap);
 	FREE_LOCK(dp->i_ump);
 }
 
 /*
  * Called whenever the link count on an inode is changed.
  * It creates an inode dependency so that the new reference(s)
  * to the inode cannot be committed to disk until the updated
  * inode has been written.
  */
 void
 softdep_change_linkcnt(ip)
 	struct inode *ip;	/* the inode with the increased link count */
 {
 	struct inodedep *inodedep;
 	int dflags;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_change_linkcnt called on non-softdep filesystem"));
 	ACQUIRE_LOCK(ip->i_ump);
 	dflags = DEPALLOC;
 	if (IS_SNAPSHOT(ip))
 		dflags |= NODELAY;
 	inodedep_lookup(UFSTOVFS(ip->i_ump), ip->i_number, dflags, &inodedep);
 	if (ip->i_nlink < ip->i_effnlink)
 		panic("softdep_change_linkcnt: bad delta");
 	inodedep->id_nlinkdelta = ip->i_nlink - ip->i_effnlink;
 	FREE_LOCK(ip->i_ump);
 }
 
 /*
  * Attach a sbdep dependency to the superblock buf so that we can keep
  * track of the head of the linked list of referenced but unlinked inodes.
  */
 void
 softdep_setup_sbupdate(ump, fs, bp)
 	struct ufsmount *ump;
 	struct fs *fs;
 	struct buf *bp;
 {
 	struct sbdep *sbdep;
 	struct worklist *wk;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ump)) != 0,
 	    ("softdep_setup_sbupdate called on non-softdep filesystem"));
 	LIST_FOREACH(wk, &bp->b_dep, wk_list)
 		if (wk->wk_type == D_SBDEP)
 			break;
 	if (wk != NULL)
 		return;
 	sbdep = malloc(sizeof(struct sbdep), M_SBDEP, M_SOFTDEP_FLAGS);
 	workitem_alloc(&sbdep->sb_list, D_SBDEP, UFSTOVFS(ump));
 	sbdep->sb_fs = fs;
 	sbdep->sb_ump = ump;
 	ACQUIRE_LOCK(ump);
 	WORKLIST_INSERT(&bp->b_dep, &sbdep->sb_list);
 	FREE_LOCK(ump);
 }
 
 /*
  * Return the first unlinked inodedep which is ready to be the head of the
  * list.  The inodedep and all those after it must have valid next pointers.
  */
 static struct inodedep *
 first_unlinked_inodedep(ump)
 	struct ufsmount *ump;
 {
 	struct inodedep *inodedep;
 	struct inodedep *idp;
 
 	LOCK_OWNED(ump);
 	for (inodedep = TAILQ_LAST(&ump->softdep_unlinked, inodedeplst);
 	    inodedep; inodedep = idp) {
 		if ((inodedep->id_state & UNLINKNEXT) == 0)
 			return (NULL);
 		idp = TAILQ_PREV(inodedep, inodedeplst, id_unlinked);
 		if (idp == NULL || (idp->id_state & UNLINKNEXT) == 0)
 			break;
 		if ((inodedep->id_state & UNLINKPREV) == 0)
 			break;
 	}
 	return (inodedep);
 }
 
 /*
  * Set the sujfree unlinked head pointer prior to writing a superblock.
  */
 static void
 initiate_write_sbdep(sbdep)
 	struct sbdep *sbdep;
 {
 	struct inodedep *inodedep;
 	struct fs *bpfs;
 	struct fs *fs;
 
 	bpfs = sbdep->sb_fs;
 	fs = sbdep->sb_ump->um_fs;
 	inodedep = first_unlinked_inodedep(sbdep->sb_ump);
 	if (inodedep) {
 		fs->fs_sujfree = inodedep->id_ino;
 		inodedep->id_state |= UNLINKPREV;
 	} else
 		fs->fs_sujfree = 0;
 	bpfs->fs_sujfree = fs->fs_sujfree;
 }
 
 /*
  * After a superblock is written determine whether it must be written again
  * due to a changing unlinked list head.
  */
 static int
 handle_written_sbdep(sbdep, bp)
 	struct sbdep *sbdep;
 	struct buf *bp;
 {
 	struct inodedep *inodedep;
 	struct mount *mp;
 	struct fs *fs;
 
 	LOCK_OWNED(sbdep->sb_ump);
 	fs = sbdep->sb_fs;
 	mp = UFSTOVFS(sbdep->sb_ump);
 	/*
 	 * If the superblock doesn't match the in-memory list start over.
 	 */
 	inodedep = first_unlinked_inodedep(sbdep->sb_ump);
 	if ((inodedep && fs->fs_sujfree != inodedep->id_ino) ||
 	    (inodedep == NULL && fs->fs_sujfree != 0)) {
 		bdirty(bp);
 		return (1);
 	}
 	WORKITEM_FREE(sbdep, D_SBDEP);
 	if (fs->fs_sujfree == 0)
 		return (0);
 	/*
 	 * Now that we have a record of this inode in stable store allow it
 	 * to be written to free up pending work.  Inodes may see a lot of
 	 * write activity after they are unlinked which we must not hold up.
 	 */
 	for (; inodedep != NULL; inodedep = TAILQ_NEXT(inodedep, id_unlinked)) {
 		if ((inodedep->id_state & UNLINKLINKS) != UNLINKLINKS)
 			panic("handle_written_sbdep: Bad inodedep %p (0x%X)",
 			    inodedep, inodedep->id_state);
 		if (inodedep->id_state & UNLINKONLIST)
 			break;
 		inodedep->id_state |= DEPCOMPLETE | UNLINKONLIST;
 	}
 
 	return (0);
 }
 
 /*
  * Mark an inodedep as unlinked and insert it into the in-memory unlinked list.
  */
 static void
 unlinked_inodedep(mp, inodedep)
 	struct mount *mp;
 	struct inodedep *inodedep;
 {
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	if (MOUNTEDSUJ(mp) == 0)
 		return;
 	ump->um_fs->fs_fmod = 1;
 	if (inodedep->id_state & UNLINKED)
 		panic("unlinked_inodedep: %p already unlinked\n", inodedep);
 	inodedep->id_state |= UNLINKED;
 	TAILQ_INSERT_HEAD(&ump->softdep_unlinked, inodedep, id_unlinked);
 }
 
 /*
  * Remove an inodedep from the unlinked inodedep list.  This may require
  * disk writes if the inode has made it that far.
  */
 static void
 clear_unlinked_inodedep(inodedep)
 	struct inodedep *inodedep;
 {
 	struct ufsmount *ump;
 	struct inodedep *idp;
 	struct inodedep *idn;
 	struct fs *fs;
 	struct buf *bp;
 	ino_t ino;
 	ino_t nino;
 	ino_t pino;
 	int error;
 
 	ump = VFSTOUFS(inodedep->id_list.wk_mp);
 	fs = ump->um_fs;
 	ino = inodedep->id_ino;
 	error = 0;
 	for (;;) {
 		LOCK_OWNED(ump);
 		KASSERT((inodedep->id_state & UNLINKED) != 0,
 		    ("clear_unlinked_inodedep: inodedep %p not unlinked",
 		    inodedep));
 		/*
 		 * If nothing has yet been written simply remove us from
 		 * the in memory list and return.  This is the most common
 		 * case where handle_workitem_remove() loses the final
 		 * reference.
 		 */
 		if ((inodedep->id_state & UNLINKLINKS) == 0)
 			break;
 		/*
 		 * If we have a NEXT pointer and no PREV pointer we can simply
 		 * clear NEXT's PREV and remove ourselves from the list.  Be
 		 * careful not to clear PREV if the superblock points at
 		 * next as well.
 		 */
 		idn = TAILQ_NEXT(inodedep, id_unlinked);
 		if ((inodedep->id_state & UNLINKLINKS) == UNLINKNEXT) {
 			if (idn && fs->fs_sujfree != idn->id_ino)
 				idn->id_state &= ~UNLINKPREV;
 			break;
 		}
 		/*
 		 * Here we have an inodedep which is actually linked into
 		 * the list.  We must remove it by forcing a write to the
 		 * link before us, whether it be the superblock or an inode.
 		 * Unfortunately the list may change while we're waiting
 		 * on the buf lock for either resource so we must loop until
 		 * we lock the right one.  If both the superblock and an
 		 * inode point to this inode we must clear the inode first
 		 * followed by the superblock.
 		 */
 		idp = TAILQ_PREV(inodedep, inodedeplst, id_unlinked);
 		pino = 0;
 		if (idp && (idp->id_state & UNLINKNEXT))
 			pino = idp->id_ino;
 		FREE_LOCK(ump);
 		if (pino == 0) {
 			bp = getblk(ump->um_devvp, btodb(fs->fs_sblockloc),
 			    (int)fs->fs_sbsize, 0, 0, 0);
 		} else {
 			error = bread(ump->um_devvp,
 			    fsbtodb(fs, ino_to_fsba(fs, pino)),
 			    (int)fs->fs_bsize, NOCRED, &bp);
 			if (error)
 				brelse(bp);
 		}
 		ACQUIRE_LOCK(ump);
 		if (error)
 			break;
 		/* If the list has changed restart the loop. */
 		idp = TAILQ_PREV(inodedep, inodedeplst, id_unlinked);
 		nino = 0;
 		if (idp && (idp->id_state & UNLINKNEXT))
 			nino = idp->id_ino;
 		if (nino != pino ||
 		    (inodedep->id_state & UNLINKPREV) != UNLINKPREV) {
 			FREE_LOCK(ump);
 			brelse(bp);
 			ACQUIRE_LOCK(ump);
 			continue;
 		}
 		nino = 0;
 		idn = TAILQ_NEXT(inodedep, id_unlinked);
 		if (idn)
 			nino = idn->id_ino;
 		/*
 		 * Remove us from the in memory list.  After this we cannot
 		 * access the inodedep.
 		 */
 		KASSERT((inodedep->id_state & UNLINKED) != 0,
 		    ("clear_unlinked_inodedep: inodedep %p not unlinked",
 		    inodedep));
 		inodedep->id_state &= ~(UNLINKED | UNLINKLINKS | UNLINKONLIST);
 		TAILQ_REMOVE(&ump->softdep_unlinked, inodedep, id_unlinked);
 		FREE_LOCK(ump);
 		/*
 		 * The predecessor's next pointer is manually updated here
 		 * so that the NEXT flag is never cleared for an element
 		 * that is in the list.
 		 */
 		if (pino == 0) {
 			bcopy((caddr_t)fs, bp->b_data, (u_int)fs->fs_sbsize);
 			ffs_oldfscompat_write((struct fs *)bp->b_data, ump);
 			softdep_setup_sbupdate(ump, (struct fs *)bp->b_data,
 			    bp);
 		} else if (fs->fs_magic == FS_UFS1_MAGIC)
 			((struct ufs1_dinode *)bp->b_data +
 			    ino_to_fsbo(fs, pino))->di_freelink = nino;
 		else
 			((struct ufs2_dinode *)bp->b_data +
 			    ino_to_fsbo(fs, pino))->di_freelink = nino;
 		/*
 		 * If the bwrite fails we have no recourse to recover.  The
 		 * filesystem is corrupted already.
 		 */
 		bwrite(bp);
 		ACQUIRE_LOCK(ump);
 		/*
 		 * If the superblock pointer still needs to be cleared force
 		 * a write here.
 		 */
 		if (fs->fs_sujfree == ino) {
 			FREE_LOCK(ump);
 			bp = getblk(ump->um_devvp, btodb(fs->fs_sblockloc),
 			    (int)fs->fs_sbsize, 0, 0, 0);
 			bcopy((caddr_t)fs, bp->b_data, (u_int)fs->fs_sbsize);
 			ffs_oldfscompat_write((struct fs *)bp->b_data, ump);
 			softdep_setup_sbupdate(ump, (struct fs *)bp->b_data,
 			    bp);
 			bwrite(bp);
 			ACQUIRE_LOCK(ump);
 		}
 
 		if (fs->fs_sujfree != ino)
 			return;
 		panic("clear_unlinked_inodedep: Failed to clear free head");
 	}
 	if (inodedep->id_ino == fs->fs_sujfree)
 		panic("clear_unlinked_inodedep: Freeing head of free list");
 	inodedep->id_state &= ~(UNLINKED | UNLINKLINKS | UNLINKONLIST);
 	TAILQ_REMOVE(&ump->softdep_unlinked, inodedep, id_unlinked);
 	return;
 }
 
 /*
  * This workitem decrements the inode's link count.
  * If the link count reaches zero, the file is removed.
  */
 static int
 handle_workitem_remove(dirrem, flags)
 	struct dirrem *dirrem;
 	int flags;
 {
 	struct inodedep *inodedep;
 	struct workhead dotdotwk;
 	struct worklist *wk;
 	struct ufsmount *ump;
 	struct mount *mp;
 	struct vnode *vp;
 	struct inode *ip;
 	ino_t oldinum;
 
 	if (dirrem->dm_state & ONWORKLIST)
 		panic("handle_workitem_remove: dirrem %p still on worklist",
 		    dirrem);
 	oldinum = dirrem->dm_oldinum;
 	mp = dirrem->dm_list.wk_mp;
 	ump = VFSTOUFS(mp);
 	flags |= LK_EXCLUSIVE;
 	if (ffs_vgetf(mp, oldinum, flags, &vp, FFSV_FORCEINSMQ) != 0)
 		return (EBUSY);
 	ip = VTOI(vp);
 	ACQUIRE_LOCK(ump);
 	if ((inodedep_lookup(mp, oldinum, 0, &inodedep)) == 0)
 		panic("handle_workitem_remove: lost inodedep");
 	if (dirrem->dm_state & ONDEPLIST)
 		LIST_REMOVE(dirrem, dm_inonext);
 	KASSERT(LIST_EMPTY(&dirrem->dm_jremrefhd),
 	    ("handle_workitem_remove:  Journal entries not written."));
 
 	/*
 	 * Move all dependencies waiting on the remove to complete
 	 * from the dirrem to the inode inowait list to be completed
 	 * after the inode has been updated and written to disk.  Any
 	 * marked MKDIR_PARENT are saved to be completed when the .. ref
 	 * is removed.
 	 */
 	LIST_INIT(&dotdotwk);
 	while ((wk = LIST_FIRST(&dirrem->dm_jwork)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		if (wk->wk_state & MKDIR_PARENT) {
 			wk->wk_state &= ~MKDIR_PARENT;
 			WORKLIST_INSERT(&dotdotwk, wk);
 			continue;
 		}
 		WORKLIST_INSERT(&inodedep->id_inowait, wk);
 	}
 	LIST_SWAP(&dirrem->dm_jwork, &dotdotwk, worklist, wk_list);
 	/*
 	 * Normal file deletion.
 	 */
 	if ((dirrem->dm_state & RMDIR) == 0) {
 		ip->i_nlink--;
 		DIP_SET(ip, i_nlink, ip->i_nlink);
 		ip->i_flag |= IN_CHANGE;
 		if (ip->i_nlink < ip->i_effnlink)
 			panic("handle_workitem_remove: bad file delta");
 		if (ip->i_nlink == 0) 
 			unlinked_inodedep(mp, inodedep);
 		inodedep->id_nlinkdelta = ip->i_nlink - ip->i_effnlink;
 		KASSERT(LIST_EMPTY(&dirrem->dm_jwork),
 		    ("handle_workitem_remove: worklist not empty. %s",
 		    TYPENAME(LIST_FIRST(&dirrem->dm_jwork)->wk_type)));
 		WORKITEM_FREE(dirrem, D_DIRREM);
 		FREE_LOCK(ump);
 		goto out;
 	}
 	/*
 	 * Directory deletion. Decrement reference count for both the
 	 * just deleted parent directory entry and the reference for ".".
 	 * Arrange to have the reference count on the parent decremented
 	 * to account for the loss of "..".
 	 */
 	ip->i_nlink -= 2;
 	DIP_SET(ip, i_nlink, ip->i_nlink);
 	ip->i_flag |= IN_CHANGE;
 	if (ip->i_nlink < ip->i_effnlink)
 		panic("handle_workitem_remove: bad dir delta");
 	if (ip->i_nlink == 0)
 		unlinked_inodedep(mp, inodedep);
 	inodedep->id_nlinkdelta = ip->i_nlink - ip->i_effnlink;
 	/*
 	 * Rename a directory to a new parent. Since, we are both deleting
 	 * and creating a new directory entry, the link count on the new
 	 * directory should not change. Thus we skip the followup dirrem.
 	 */
 	if (dirrem->dm_state & DIRCHG) {
 		KASSERT(LIST_EMPTY(&dirrem->dm_jwork),
 		    ("handle_workitem_remove: DIRCHG and worklist not empty."));
 		WORKITEM_FREE(dirrem, D_DIRREM);
 		FREE_LOCK(ump);
 		goto out;
 	}
 	dirrem->dm_state = ONDEPLIST;
 	dirrem->dm_oldinum = dirrem->dm_dirinum;
 	/*
 	 * Place the dirrem on the parent's diremhd list.
 	 */
 	if (inodedep_lookup(mp, dirrem->dm_oldinum, 0, &inodedep) == 0)
 		panic("handle_workitem_remove: lost dir inodedep");
 	LIST_INSERT_HEAD(&inodedep->id_dirremhd, dirrem, dm_inonext);
 	/*
 	 * If the allocated inode has never been written to disk, then
 	 * the on-disk inode is zero'ed and we can remove the file
 	 * immediately.  When journaling if the inode has been marked
 	 * unlinked and not DEPCOMPLETE we know it can never be written.
 	 */
 	inodedep_lookup(mp, oldinum, 0, &inodedep);
 	if (inodedep == NULL ||
 	    (inodedep->id_state & (DEPCOMPLETE | UNLINKED)) == UNLINKED ||
 	    check_inode_unwritten(inodedep)) {
 		FREE_LOCK(ump);
 		vput(vp);
 		return handle_workitem_remove(dirrem, flags);
 	}
 	WORKLIST_INSERT(&inodedep->id_inowait, &dirrem->dm_list);
 	FREE_LOCK(ump);
 	ip->i_flag |= IN_CHANGE;
 out:
 	ffs_update(vp, 0);
 	vput(vp);
 	return (0);
 }
 
 /*
  * Inode de-allocation dependencies.
  * 
  * When an inode's link count is reduced to zero, it can be de-allocated. We
  * found it convenient to postpone de-allocation until after the inode is
  * written to disk with its new link count (zero).  At this point, all of the
  * on-disk inode's block pointers are nullified and, with careful dependency
  * list ordering, all dependencies related to the inode will be satisfied and
  * the corresponding dependency structures de-allocated.  So, if/when the
  * inode is reused, there will be no mixing of old dependencies with new
  * ones.  This artificial dependency is set up by the block de-allocation
  * procedure above (softdep_setup_freeblocks) and completed by the
  * following procedure.
  */
 static void 
 handle_workitem_freefile(freefile)
 	struct freefile *freefile;
 {
 	struct workhead wkhd;
 	struct fs *fs;
 	struct inodedep *idp;
 	struct ufsmount *ump;
 	int error;
 
 	ump = VFSTOUFS(freefile->fx_list.wk_mp);
 	fs = ump->um_fs;
 #ifdef DEBUG
 	ACQUIRE_LOCK(ump);
 	error = inodedep_lookup(UFSTOVFS(ump), freefile->fx_oldinum, 0, &idp);
 	FREE_LOCK(ump);
 	if (error)
 		panic("handle_workitem_freefile: inodedep %p survived", idp);
 #endif
 	UFS_LOCK(ump);
 	fs->fs_pendinginodes -= 1;
 	UFS_UNLOCK(ump);
 	LIST_INIT(&wkhd);
 	LIST_SWAP(&freefile->fx_jwork, &wkhd, worklist, wk_list);
 	if ((error = ffs_freefile(ump, fs, freefile->fx_devvp,
 	    freefile->fx_oldinum, freefile->fx_mode, &wkhd)) != 0)
 		softdep_error("handle_workitem_freefile", error);
 	ACQUIRE_LOCK(ump);
 	WORKITEM_FREE(freefile, D_FREEFILE);
 	FREE_LOCK(ump);
 }
 
 
 /*
  * Helper function which unlinks marker element from work list and returns
  * the next element on the list.
  */
 static __inline struct worklist *
 markernext(struct worklist *marker)
 {
 	struct worklist *next;
 	
 	next = LIST_NEXT(marker, wk_list);
 	LIST_REMOVE(marker, wk_list);
 	return next;
 }
 
 /*
  * Disk writes.
  * 
  * The dependency structures constructed above are most actively used when file
  * system blocks are written to disk.  No constraints are placed on when a
  * block can be written, but unsatisfied update dependencies are made safe by
  * modifying (or replacing) the source memory for the duration of the disk
  * write.  When the disk write completes, the memory block is again brought
  * up-to-date.
  *
  * In-core inode structure reclamation.
  * 
  * Because there are a finite number of "in-core" inode structures, they are
  * reused regularly.  By transferring all inode-related dependencies to the
  * in-memory inode block and indexing them separately (via "inodedep"s), we
  * can allow "in-core" inode structures to be reused at any time and avoid
  * any increase in contention.
  *
  * Called just before entering the device driver to initiate a new disk I/O.
  * The buffer must be locked, thus, no I/O completion operations can occur
  * while we are manipulating its associated dependencies.
  */
 static void 
 softdep_disk_io_initiation(bp)
 	struct buf *bp;		/* structure describing disk write to occur */
 {
 	struct worklist *wk;
 	struct worklist marker;
 	struct inodedep *inodedep;
 	struct freeblks *freeblks;
 	struct jblkdep *jblkdep;
 	struct newblk *newblk;
 	struct ufsmount *ump;
 
 	/*
 	 * We only care about write operations. There should never
 	 * be dependencies for reads.
 	 */
 	if (bp->b_iocmd != BIO_WRITE)
 		panic("softdep_disk_io_initiation: not write");
 
 	if (bp->b_vflags & BV_BKGRDINPROG)
 		panic("softdep_disk_io_initiation: Writing buffer with "
 		    "background write in progress: %p", bp);
 
 	if ((wk = LIST_FIRST(&bp->b_dep)) == NULL)
 		return;
 	ump = VFSTOUFS(wk->wk_mp);
 
 	marker.wk_type = D_LAST + 1;	/* Not a normal workitem */
 	PHOLD(curproc);			/* Don't swap out kernel stack */
 	ACQUIRE_LOCK(ump);
 	/*
 	 * Do any necessary pre-I/O processing.
 	 */
 	for (wk = LIST_FIRST(&bp->b_dep); wk != NULL;
 	     wk = markernext(&marker)) {
 		LIST_INSERT_AFTER(wk, &marker, wk_list);
 		switch (wk->wk_type) {
 
 		case D_PAGEDEP:
 			initiate_write_filepage(WK_PAGEDEP(wk), bp);
 			continue;
 
 		case D_INODEDEP:
 			inodedep = WK_INODEDEP(wk);
 			if (inodedep->id_fs->fs_magic == FS_UFS1_MAGIC)
 				initiate_write_inodeblock_ufs1(inodedep, bp);
 			else
 				initiate_write_inodeblock_ufs2(inodedep, bp);
 			continue;
 
 		case D_INDIRDEP:
 			initiate_write_indirdep(WK_INDIRDEP(wk), bp);
 			continue;
 
 		case D_BMSAFEMAP:
 			initiate_write_bmsafemap(WK_BMSAFEMAP(wk), bp);
 			continue;
 
 		case D_JSEG:
 			WK_JSEG(wk)->js_buf = NULL;
 			continue;
 
 		case D_FREEBLKS:
 			freeblks = WK_FREEBLKS(wk);
 			jblkdep = LIST_FIRST(&freeblks->fb_jblkdephd);
 			/*
 			 * We have to wait for the freeblks to be journaled
 			 * before we can write an inodeblock with updated
 			 * pointers.  Be careful to arrange the marker so
 			 * we revisit the freeblks if it's not removed by
 			 * the first jwait().
 			 */
 			if (jblkdep != NULL) {
 				LIST_REMOVE(&marker, wk_list);
 				LIST_INSERT_BEFORE(wk, &marker, wk_list);
 				jwait(&jblkdep->jb_list, MNT_WAIT);
 			}
 			continue;
 		case D_ALLOCDIRECT:
 		case D_ALLOCINDIR:
 			/*
 			 * We have to wait for the jnewblk to be journaled
 			 * before we can write to a block if the contents
 			 * may be confused with an earlier file's indirect
 			 * at recovery time.  Handle the marker as described
 			 * above.
 			 */
 			newblk = WK_NEWBLK(wk);
 			if (newblk->nb_jnewblk != NULL &&
 			    indirblk_lookup(newblk->nb_list.wk_mp,
 			    newblk->nb_newblkno)) {
 				LIST_REMOVE(&marker, wk_list);
 				LIST_INSERT_BEFORE(wk, &marker, wk_list);
 				jwait(&newblk->nb_jnewblk->jn_list, MNT_WAIT);
 			}
 			continue;
 
 		case D_SBDEP:
 			initiate_write_sbdep(WK_SBDEP(wk));
 			continue;
 
 		case D_MKDIR:
 		case D_FREEWORK:
 		case D_FREEDEP:
 		case D_JSEGDEP:
 			continue;
 
 		default:
 			panic("handle_disk_io_initiation: Unexpected type %s",
 			    TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 	}
 	FREE_LOCK(ump);
 	PRELE(curproc);			/* Allow swapout of kernel stack */
 }
 
 /*
  * Called from within the procedure above to deal with unsatisfied
  * allocation dependencies in a directory. The buffer must be locked,
  * thus, no I/O completion operations can occur while we are
  * manipulating its associated dependencies.
  */
 static void
 initiate_write_filepage(pagedep, bp)
 	struct pagedep *pagedep;
 	struct buf *bp;
 {
 	struct jremref *jremref;
 	struct jmvref *jmvref;
 	struct dirrem *dirrem;
 	struct diradd *dap;
 	struct direct *ep;
 	int i;
 
 	if (pagedep->pd_state & IOSTARTED) {
 		/*
 		 * This can only happen if there is a driver that does not
 		 * understand chaining. Here biodone will reissue the call
 		 * to strategy for the incomplete buffers.
 		 */
 		printf("initiate_write_filepage: already started\n");
 		return;
 	}
 	pagedep->pd_state |= IOSTARTED;
 	/*
 	 * Wait for all journal remove dependencies to hit the disk.
 	 * We can not allow any potentially conflicting directory adds
 	 * to be visible before removes and rollback is too difficult.
 	 * The per-filesystem lock may be dropped and re-acquired, however 
 	 * we hold the buf locked so the dependency can not go away.
 	 */
 	LIST_FOREACH(dirrem, &pagedep->pd_dirremhd, dm_next)
 		while ((jremref = LIST_FIRST(&dirrem->dm_jremrefhd)) != NULL)
 			jwait(&jremref->jr_list, MNT_WAIT);
 	while ((jmvref = LIST_FIRST(&pagedep->pd_jmvrefhd)) != NULL)
 		jwait(&jmvref->jm_list, MNT_WAIT);
 	for (i = 0; i < DAHASHSZ; i++) {
 		LIST_FOREACH(dap, &pagedep->pd_diraddhd[i], da_pdlist) {
 			ep = (struct direct *)
 			    ((char *)bp->b_data + dap->da_offset);
 			if (ep->d_ino != dap->da_newinum)
 				panic("%s: dir inum %ju != new %ju",
 				    "initiate_write_filepage",
 				    (uintmax_t)ep->d_ino,
 				    (uintmax_t)dap->da_newinum);
 			if (dap->da_state & DIRCHG)
 				ep->d_ino = dap->da_previous->dm_oldinum;
 			else
 				ep->d_ino = 0;
 			dap->da_state &= ~ATTACHED;
 			dap->da_state |= UNDONE;
 		}
 	}
 }
 
 /*
  * Version of initiate_write_inodeblock that handles UFS1 dinodes.
  * Note that any bug fixes made to this routine must be done in the
  * version found below.
  *
  * Called from within the procedure above to deal with unsatisfied
  * allocation dependencies in an inodeblock. The buffer must be
  * locked, thus, no I/O completion operations can occur while we
  * are manipulating its associated dependencies.
  */
 static void 
 initiate_write_inodeblock_ufs1(inodedep, bp)
 	struct inodedep *inodedep;
 	struct buf *bp;			/* The inode block */
 {
 	struct allocdirect *adp, *lastadp;
 	struct ufs1_dinode *dp;
 	struct ufs1_dinode *sip;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	struct fs *fs;
 	ufs_lbn_t i;
 #ifdef INVARIANTS
 	ufs_lbn_t prevlbn = 0;
 #endif
 	int deplist;
 
 	if (inodedep->id_state & IOSTARTED)
 		panic("initiate_write_inodeblock_ufs1: already started");
 	inodedep->id_state |= IOSTARTED;
 	fs = inodedep->id_fs;
 	ump = VFSTOUFS(inodedep->id_list.wk_mp);
 	LOCK_OWNED(ump);
 	dp = (struct ufs1_dinode *)bp->b_data +
 	    ino_to_fsbo(fs, inodedep->id_ino);
 
 	/*
 	 * If we're on the unlinked list but have not yet written our
 	 * next pointer initialize it here.
 	 */
 	if ((inodedep->id_state & (UNLINKED | UNLINKNEXT)) == UNLINKED) {
 		struct inodedep *inon;
 
 		inon = TAILQ_NEXT(inodedep, id_unlinked);
 		dp->di_freelink = inon ? inon->id_ino : 0;
 	}
 	/*
 	 * If the bitmap is not yet written, then the allocated
 	 * inode cannot be written to disk.
 	 */
 	if ((inodedep->id_state & DEPCOMPLETE) == 0) {
 		if (inodedep->id_savedino1 != NULL)
 			panic("initiate_write_inodeblock_ufs1: I/O underway");
 		FREE_LOCK(ump);
 		sip = malloc(sizeof(struct ufs1_dinode),
 		    M_SAVEDINO, M_SOFTDEP_FLAGS);
 		ACQUIRE_LOCK(ump);
 		inodedep->id_savedino1 = sip;
 		*inodedep->id_savedino1 = *dp;
 		bzero((caddr_t)dp, sizeof(struct ufs1_dinode));
 		dp->di_gen = inodedep->id_savedino1->di_gen;
 		dp->di_freelink = inodedep->id_savedino1->di_freelink;
 		return;
 	}
 	/*
 	 * If no dependencies, then there is nothing to roll back.
 	 */
 	inodedep->id_savedsize = dp->di_size;
 	inodedep->id_savedextsize = 0;
 	inodedep->id_savednlink = dp->di_nlink;
 	if (TAILQ_EMPTY(&inodedep->id_inoupdt) &&
 	    TAILQ_EMPTY(&inodedep->id_inoreflst))
 		return;
 	/*
 	 * Revert the link count to that of the first unwritten journal entry.
 	 */
 	inoref = TAILQ_FIRST(&inodedep->id_inoreflst);
 	if (inoref)
 		dp->di_nlink = inoref->if_nlink;
 	/*
 	 * Set the dependencies to busy.
 	 */
 	for (deplist = 0, adp = TAILQ_FIRST(&inodedep->id_inoupdt); adp;
 	     adp = TAILQ_NEXT(adp, ad_next)) {
 #ifdef INVARIANTS
 		if (deplist != 0 && prevlbn >= adp->ad_offset)
 			panic("softdep_write_inodeblock: lbn order");
 		prevlbn = adp->ad_offset;
 		if (adp->ad_offset < NDADDR &&
 		    dp->di_db[adp->ad_offset] != adp->ad_newblkno)
 			panic("%s: direct pointer #%jd mismatch %d != %jd",
 			    "softdep_write_inodeblock",
 			    (intmax_t)adp->ad_offset,
 			    dp->di_db[adp->ad_offset],
 			    (intmax_t)adp->ad_newblkno);
 		if (adp->ad_offset >= NDADDR &&
 		    dp->di_ib[adp->ad_offset - NDADDR] != adp->ad_newblkno)
 			panic("%s: indirect pointer #%jd mismatch %d != %jd",
 			    "softdep_write_inodeblock",
 			    (intmax_t)adp->ad_offset - NDADDR,
 			    dp->di_ib[adp->ad_offset - NDADDR],
 			    (intmax_t)adp->ad_newblkno);
 		deplist |= 1 << adp->ad_offset;
 		if ((adp->ad_state & ATTACHED) == 0)
 			panic("softdep_write_inodeblock: Unknown state 0x%x",
 			    adp->ad_state);
 #endif /* INVARIANTS */
 		adp->ad_state &= ~ATTACHED;
 		adp->ad_state |= UNDONE;
 	}
 	/*
 	 * The on-disk inode cannot claim to be any larger than the last
 	 * fragment that has been written. Otherwise, the on-disk inode
 	 * might have fragments that were not the last block in the file
 	 * which would corrupt the filesystem.
 	 */
 	for (lastadp = NULL, adp = TAILQ_FIRST(&inodedep->id_inoupdt); adp;
 	     lastadp = adp, adp = TAILQ_NEXT(adp, ad_next)) {
 		if (adp->ad_offset >= NDADDR)
 			break;
 		dp->di_db[adp->ad_offset] = adp->ad_oldblkno;
 		/* keep going until hitting a rollback to a frag */
 		if (adp->ad_oldsize == 0 || adp->ad_oldsize == fs->fs_bsize)
 			continue;
 		dp->di_size = fs->fs_bsize * adp->ad_offset + adp->ad_oldsize;
 		for (i = adp->ad_offset + 1; i < NDADDR; i++) {
 #ifdef INVARIANTS
 			if (dp->di_db[i] != 0 && (deplist & (1 << i)) == 0)
 				panic("softdep_write_inodeblock: lost dep1");
 #endif /* INVARIANTS */
 			dp->di_db[i] = 0;
 		}
 		for (i = 0; i < NIADDR; i++) {
 #ifdef INVARIANTS
 			if (dp->di_ib[i] != 0 &&
 			    (deplist & ((1 << NDADDR) << i)) == 0)
 				panic("softdep_write_inodeblock: lost dep2");
 #endif /* INVARIANTS */
 			dp->di_ib[i] = 0;
 		}
 		return;
 	}
 	/*
 	 * If we have zero'ed out the last allocated block of the file,
 	 * roll back the size to the last currently allocated block.
 	 * We know that this last allocated block is a full-sized as
 	 * we already checked for fragments in the loop above.
 	 */
 	if (lastadp != NULL &&
 	    dp->di_size <= (lastadp->ad_offset + 1) * fs->fs_bsize) {
 		for (i = lastadp->ad_offset; i >= 0; i--)
 			if (dp->di_db[i] != 0)
 				break;
 		dp->di_size = (i + 1) * fs->fs_bsize;
 	}
 	/*
 	 * The only dependencies are for indirect blocks.
 	 *
 	 * The file size for indirect block additions is not guaranteed.
 	 * Such a guarantee would be non-trivial to achieve. The conventional
 	 * synchronous write implementation also does not make this guarantee.
 	 * Fsck should catch and fix discrepancies. Arguably, the file size
 	 * can be over-estimated without destroying integrity when the file
 	 * moves into the indirect blocks (i.e., is large). If we want to
 	 * postpone fsck, we are stuck with this argument.
 	 */
 	for (; adp; adp = TAILQ_NEXT(adp, ad_next))
 		dp->di_ib[adp->ad_offset - NDADDR] = 0;
 }
 		
 /*
  * Version of initiate_write_inodeblock that handles UFS2 dinodes.
  * Note that any bug fixes made to this routine must be done in the
  * version found above.
  *
  * Called from within the procedure above to deal with unsatisfied
  * allocation dependencies in an inodeblock. The buffer must be
  * locked, thus, no I/O completion operations can occur while we
  * are manipulating its associated dependencies.
  */
 static void 
 initiate_write_inodeblock_ufs2(inodedep, bp)
 	struct inodedep *inodedep;
 	struct buf *bp;			/* The inode block */
 {
 	struct allocdirect *adp, *lastadp;
 	struct ufs2_dinode *dp;
 	struct ufs2_dinode *sip;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	struct fs *fs;
 	ufs_lbn_t i;
 #ifdef INVARIANTS
 	ufs_lbn_t prevlbn = 0;
 #endif
 	int deplist;
 
 	if (inodedep->id_state & IOSTARTED)
 		panic("initiate_write_inodeblock_ufs2: already started");
 	inodedep->id_state |= IOSTARTED;
 	fs = inodedep->id_fs;
 	ump = VFSTOUFS(inodedep->id_list.wk_mp);
 	LOCK_OWNED(ump);
 	dp = (struct ufs2_dinode *)bp->b_data +
 	    ino_to_fsbo(fs, inodedep->id_ino);
 
 	/*
 	 * If we're on the unlinked list but have not yet written our
 	 * next pointer initialize it here.
 	 */
 	if ((inodedep->id_state & (UNLINKED | UNLINKNEXT)) == UNLINKED) {
 		struct inodedep *inon;
 
 		inon = TAILQ_NEXT(inodedep, id_unlinked);
 		dp->di_freelink = inon ? inon->id_ino : 0;
 	}
 	/*
 	 * If the bitmap is not yet written, then the allocated
 	 * inode cannot be written to disk.
 	 */
 	if ((inodedep->id_state & DEPCOMPLETE) == 0) {
 		if (inodedep->id_savedino2 != NULL)
 			panic("initiate_write_inodeblock_ufs2: I/O underway");
 		FREE_LOCK(ump);
 		sip = malloc(sizeof(struct ufs2_dinode),
 		    M_SAVEDINO, M_SOFTDEP_FLAGS);
 		ACQUIRE_LOCK(ump);
 		inodedep->id_savedino2 = sip;
 		*inodedep->id_savedino2 = *dp;
 		bzero((caddr_t)dp, sizeof(struct ufs2_dinode));
 		dp->di_gen = inodedep->id_savedino2->di_gen;
 		dp->di_freelink = inodedep->id_savedino2->di_freelink;
 		return;
 	}
 	/*
 	 * If no dependencies, then there is nothing to roll back.
 	 */
 	inodedep->id_savedsize = dp->di_size;
 	inodedep->id_savedextsize = dp->di_extsize;
 	inodedep->id_savednlink = dp->di_nlink;
 	if (TAILQ_EMPTY(&inodedep->id_inoupdt) &&
 	    TAILQ_EMPTY(&inodedep->id_extupdt) &&
 	    TAILQ_EMPTY(&inodedep->id_inoreflst))
 		return;
 	/*
 	 * Revert the link count to that of the first unwritten journal entry.
 	 */
 	inoref = TAILQ_FIRST(&inodedep->id_inoreflst);
 	if (inoref)
 		dp->di_nlink = inoref->if_nlink;
 
 	/*
 	 * Set the ext data dependencies to busy.
 	 */
 	for (deplist = 0, adp = TAILQ_FIRST(&inodedep->id_extupdt); adp;
 	     adp = TAILQ_NEXT(adp, ad_next)) {
 #ifdef INVARIANTS
 		if (deplist != 0 && prevlbn >= adp->ad_offset)
 			panic("softdep_write_inodeblock: lbn order");
 		prevlbn = adp->ad_offset;
 		if (dp->di_extb[adp->ad_offset] != adp->ad_newblkno)
 			panic("%s: direct pointer #%jd mismatch %jd != %jd",
 			    "softdep_write_inodeblock",
 			    (intmax_t)adp->ad_offset,
 			    (intmax_t)dp->di_extb[adp->ad_offset],
 			    (intmax_t)adp->ad_newblkno);
 		deplist |= 1 << adp->ad_offset;
 		if ((adp->ad_state & ATTACHED) == 0)
 			panic("softdep_write_inodeblock: Unknown state 0x%x",
 			    adp->ad_state);
 #endif /* INVARIANTS */
 		adp->ad_state &= ~ATTACHED;
 		adp->ad_state |= UNDONE;
 	}
 	/*
 	 * The on-disk inode cannot claim to be any larger than the last
 	 * fragment that has been written. Otherwise, the on-disk inode
 	 * might have fragments that were not the last block in the ext
 	 * data which would corrupt the filesystem.
 	 */
 	for (lastadp = NULL, adp = TAILQ_FIRST(&inodedep->id_extupdt); adp;
 	     lastadp = adp, adp = TAILQ_NEXT(adp, ad_next)) {
 		dp->di_extb[adp->ad_offset] = adp->ad_oldblkno;
 		/* keep going until hitting a rollback to a frag */
 		if (adp->ad_oldsize == 0 || adp->ad_oldsize == fs->fs_bsize)
 			continue;
 		dp->di_extsize = fs->fs_bsize * adp->ad_offset + adp->ad_oldsize;
 		for (i = adp->ad_offset + 1; i < NXADDR; i++) {
 #ifdef INVARIANTS
 			if (dp->di_extb[i] != 0 && (deplist & (1 << i)) == 0)
 				panic("softdep_write_inodeblock: lost dep1");
 #endif /* INVARIANTS */
 			dp->di_extb[i] = 0;
 		}
 		lastadp = NULL;
 		break;
 	}
 	/*
 	 * If we have zero'ed out the last allocated block of the ext
 	 * data, roll back the size to the last currently allocated block.
 	 * We know that this last allocated block is a full-sized as
 	 * we already checked for fragments in the loop above.
 	 */
 	if (lastadp != NULL &&
 	    dp->di_extsize <= (lastadp->ad_offset + 1) * fs->fs_bsize) {
 		for (i = lastadp->ad_offset; i >= 0; i--)
 			if (dp->di_extb[i] != 0)
 				break;
 		dp->di_extsize = (i + 1) * fs->fs_bsize;
 	}
 	/*
 	 * Set the file data dependencies to busy.
 	 */
 	for (deplist = 0, adp = TAILQ_FIRST(&inodedep->id_inoupdt); adp;
 	     adp = TAILQ_NEXT(adp, ad_next)) {
 #ifdef INVARIANTS
 		if (deplist != 0 && prevlbn >= adp->ad_offset)
 			panic("softdep_write_inodeblock: lbn order");
 		if ((adp->ad_state & ATTACHED) == 0)
 			panic("inodedep %p and adp %p not attached", inodedep, adp);
 		prevlbn = adp->ad_offset;
 		if (adp->ad_offset < NDADDR &&
 		    dp->di_db[adp->ad_offset] != adp->ad_newblkno)
 			panic("%s: direct pointer #%jd mismatch %jd != %jd",
 			    "softdep_write_inodeblock",
 			    (intmax_t)adp->ad_offset,
 			    (intmax_t)dp->di_db[adp->ad_offset],
 			    (intmax_t)adp->ad_newblkno);
 		if (adp->ad_offset >= NDADDR &&
 		    dp->di_ib[adp->ad_offset - NDADDR] != adp->ad_newblkno)
 			panic("%s indirect pointer #%jd mismatch %jd != %jd",
 			    "softdep_write_inodeblock:",
 			    (intmax_t)adp->ad_offset - NDADDR,
 			    (intmax_t)dp->di_ib[adp->ad_offset - NDADDR],
 			    (intmax_t)adp->ad_newblkno);
 		deplist |= 1 << adp->ad_offset;
 		if ((adp->ad_state & ATTACHED) == 0)
 			panic("softdep_write_inodeblock: Unknown state 0x%x",
 			    adp->ad_state);
 #endif /* INVARIANTS */
 		adp->ad_state &= ~ATTACHED;
 		adp->ad_state |= UNDONE;
 	}
 	/*
 	 * The on-disk inode cannot claim to be any larger than the last
 	 * fragment that has been written. Otherwise, the on-disk inode
 	 * might have fragments that were not the last block in the file
 	 * which would corrupt the filesystem.
 	 */
 	for (lastadp = NULL, adp = TAILQ_FIRST(&inodedep->id_inoupdt); adp;
 	     lastadp = adp, adp = TAILQ_NEXT(adp, ad_next)) {
 		if (adp->ad_offset >= NDADDR)
 			break;
 		dp->di_db[adp->ad_offset] = adp->ad_oldblkno;
 		/* keep going until hitting a rollback to a frag */
 		if (adp->ad_oldsize == 0 || adp->ad_oldsize == fs->fs_bsize)
 			continue;
 		dp->di_size = fs->fs_bsize * adp->ad_offset + adp->ad_oldsize;
 		for (i = adp->ad_offset + 1; i < NDADDR; i++) {
 #ifdef INVARIANTS
 			if (dp->di_db[i] != 0 && (deplist & (1 << i)) == 0)
 				panic("softdep_write_inodeblock: lost dep2");
 #endif /* INVARIANTS */
 			dp->di_db[i] = 0;
 		}
 		for (i = 0; i < NIADDR; i++) {
 #ifdef INVARIANTS
 			if (dp->di_ib[i] != 0 &&
 			    (deplist & ((1 << NDADDR) << i)) == 0)
 				panic("softdep_write_inodeblock: lost dep3");
 #endif /* INVARIANTS */
 			dp->di_ib[i] = 0;
 		}
 		return;
 	}
 	/*
 	 * If we have zero'ed out the last allocated block of the file,
 	 * roll back the size to the last currently allocated block.
 	 * We know that this last allocated block is a full-sized as
 	 * we already checked for fragments in the loop above.
 	 */
 	if (lastadp != NULL &&
 	    dp->di_size <= (lastadp->ad_offset + 1) * fs->fs_bsize) {
 		for (i = lastadp->ad_offset; i >= 0; i--)
 			if (dp->di_db[i] != 0)
 				break;
 		dp->di_size = (i + 1) * fs->fs_bsize;
 	}
 	/*
 	 * The only dependencies are for indirect blocks.
 	 *
 	 * The file size for indirect block additions is not guaranteed.
 	 * Such a guarantee would be non-trivial to achieve. The conventional
 	 * synchronous write implementation also does not make this guarantee.
 	 * Fsck should catch and fix discrepancies. Arguably, the file size
 	 * can be over-estimated without destroying integrity when the file
 	 * moves into the indirect blocks (i.e., is large). If we want to
 	 * postpone fsck, we are stuck with this argument.
 	 */
 	for (; adp; adp = TAILQ_NEXT(adp, ad_next))
 		dp->di_ib[adp->ad_offset - NDADDR] = 0;
 }
 
 /*
  * Cancel an indirdep as a result of truncation.  Release all of the
  * children allocindirs and place their journal work on the appropriate
  * list.
  */
 static void
 cancel_indirdep(indirdep, bp, freeblks)
 	struct indirdep *indirdep;
 	struct buf *bp;
 	struct freeblks *freeblks;
 {
 	struct allocindir *aip;
 
 	/*
 	 * None of the indirect pointers will ever be visible,
 	 * so they can simply be tossed. GOINGAWAY ensures
 	 * that allocated pointers will be saved in the buffer
 	 * cache until they are freed. Note that they will
 	 * only be able to be found by their physical address
 	 * since the inode mapping the logical address will
 	 * be gone. The save buffer used for the safe copy
 	 * was allocated in setup_allocindir_phase2 using
 	 * the physical address so it could be used for this
 	 * purpose. Hence we swap the safe copy with the real
 	 * copy, allowing the safe copy to be freed and holding
 	 * on to the real copy for later use in indir_trunc.
 	 */
 	if (indirdep->ir_state & GOINGAWAY)
 		panic("cancel_indirdep: already gone");
 	if ((indirdep->ir_state & DEPCOMPLETE) == 0) {
 		indirdep->ir_state |= DEPCOMPLETE;
 		LIST_REMOVE(indirdep, ir_next);
 	}
 	indirdep->ir_state |= GOINGAWAY;
 	/*
 	 * Pass in bp for blocks still have journal writes
 	 * pending so we can cancel them on their own.
 	 */
 	while ((aip = LIST_FIRST(&indirdep->ir_deplisthd)) != 0)
 		cancel_allocindir(aip, bp, freeblks, 0);
 	while ((aip = LIST_FIRST(&indirdep->ir_donehd)) != 0)
 		cancel_allocindir(aip, NULL, freeblks, 0);
 	while ((aip = LIST_FIRST(&indirdep->ir_writehd)) != 0)
 		cancel_allocindir(aip, NULL, freeblks, 0);
 	while ((aip = LIST_FIRST(&indirdep->ir_completehd)) != 0)
 		cancel_allocindir(aip, NULL, freeblks, 0);
 	/*
 	 * If there are pending partial truncations we need to keep the
 	 * old block copy around until they complete.  This is because
 	 * the current b_data is not a perfect superset of the available
 	 * blocks.
 	 */
 	if (TAILQ_EMPTY(&indirdep->ir_trunc))
 		bcopy(bp->b_data, indirdep->ir_savebp->b_data, bp->b_bcount);
 	else
 		bcopy(bp->b_data, indirdep->ir_saveddata, bp->b_bcount);
 	WORKLIST_REMOVE(&indirdep->ir_list);
 	WORKLIST_INSERT(&indirdep->ir_savebp->b_dep, &indirdep->ir_list);
 	indirdep->ir_bp = NULL;
 	indirdep->ir_freeblks = freeblks;
 }
 
 /*
  * Free an indirdep once it no longer has new pointers to track.
  */
 static void
 free_indirdep(indirdep)
 	struct indirdep *indirdep;
 {
 
 	KASSERT(TAILQ_EMPTY(&indirdep->ir_trunc),
 	    ("free_indirdep: Indir trunc list not empty."));
 	KASSERT(LIST_EMPTY(&indirdep->ir_completehd),
 	    ("free_indirdep: Complete head not empty."));
 	KASSERT(LIST_EMPTY(&indirdep->ir_writehd),
 	    ("free_indirdep: write head not empty."));
 	KASSERT(LIST_EMPTY(&indirdep->ir_donehd),
 	    ("free_indirdep: done head not empty."));
 	KASSERT(LIST_EMPTY(&indirdep->ir_deplisthd),
 	    ("free_indirdep: deplist head not empty."));
 	KASSERT((indirdep->ir_state & DEPCOMPLETE),
 	    ("free_indirdep: %p still on newblk list.", indirdep));
 	KASSERT(indirdep->ir_saveddata == NULL,
 	    ("free_indirdep: %p still has saved data.", indirdep));
 	if (indirdep->ir_state & ONWORKLIST)
 		WORKLIST_REMOVE(&indirdep->ir_list);
 	WORKITEM_FREE(indirdep, D_INDIRDEP);
 }
 
 /*
  * Called before a write to an indirdep.  This routine is responsible for
  * rolling back pointers to a safe state which includes only those
  * allocindirs which have been completed.
  */
 static void
 initiate_write_indirdep(indirdep, bp)
 	struct indirdep *indirdep;
 	struct buf *bp;
 {
 	struct ufsmount *ump;
 
 	indirdep->ir_state |= IOSTARTED;
 	if (indirdep->ir_state & GOINGAWAY)
 		panic("disk_io_initiation: indirdep gone");
 	/*
 	 * If there are no remaining dependencies, this will be writing
 	 * the real pointers.
 	 */
 	if (LIST_EMPTY(&indirdep->ir_deplisthd) &&
 	    TAILQ_EMPTY(&indirdep->ir_trunc))
 		return;
 	/*
 	 * Replace up-to-date version with safe version.
 	 */
 	if (indirdep->ir_saveddata == NULL) {
 		ump = VFSTOUFS(indirdep->ir_list.wk_mp);
 		LOCK_OWNED(ump);
 		FREE_LOCK(ump);
 		indirdep->ir_saveddata = malloc(bp->b_bcount, M_INDIRDEP,
 		    M_SOFTDEP_FLAGS);
 		ACQUIRE_LOCK(ump);
 	}
 	indirdep->ir_state &= ~ATTACHED;
 	indirdep->ir_state |= UNDONE;
 	bcopy(bp->b_data, indirdep->ir_saveddata, bp->b_bcount);
 	bcopy(indirdep->ir_savebp->b_data, bp->b_data,
 	    bp->b_bcount);
 }
 
 /*
  * Called when an inode has been cleared in a cg bitmap.  This finally
  * eliminates any canceled jaddrefs
  */
 void
 softdep_setup_inofree(mp, bp, ino, wkhd)
 	struct mount *mp;
 	struct buf *bp;
 	ino_t ino;
 	struct workhead *wkhd;
 {
 	struct worklist *wk, *wkn;
 	struct inodedep *inodedep;
 	struct ufsmount *ump;
 	uint8_t *inosused;
 	struct cg *cgp;
 	struct fs *fs;
 
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_setup_inofree called on non-softdep filesystem"));
 	ump = VFSTOUFS(mp);
 	ACQUIRE_LOCK(ump);
 	fs = ump->um_fs;
 	cgp = (struct cg *)bp->b_data;
 	inosused = cg_inosused(cgp);
 	if (isset(inosused, ino % fs->fs_ipg))
 		panic("softdep_setup_inofree: inode %ju not freed.",
 		    (uintmax_t)ino);
 	if (inodedep_lookup(mp, ino, 0, &inodedep))
 		panic("softdep_setup_inofree: ino %ju has existing inodedep %p",
 		    (uintmax_t)ino, inodedep);
 	if (wkhd) {
 		LIST_FOREACH_SAFE(wk, wkhd, wk_list, wkn) {
 			if (wk->wk_type != D_JADDREF)
 				continue;
 			WORKLIST_REMOVE(wk);
 			/*
 			 * We can free immediately even if the jaddref
 			 * isn't attached in a background write as now
 			 * the bitmaps are reconciled.
 			 */
 			wk->wk_state |= COMPLETE | ATTACHED;
 			free_jaddref(WK_JADDREF(wk));
 		}
 		jwork_move(&bp->b_dep, wkhd);
 	}
 	FREE_LOCK(ump);
 }
 
 
 /*
  * Called via ffs_blkfree() after a set of frags has been cleared from a cg
  * map.  Any dependencies waiting for the write to clear are added to the
  * buf's list and any jnewblks that are being canceled are discarded
  * immediately.
  */
 void
 softdep_setup_blkfree(mp, bp, blkno, frags, wkhd)
 	struct mount *mp;
 	struct buf *bp;
 	ufs2_daddr_t blkno;
 	int frags;
 	struct workhead *wkhd;
 {
 	struct bmsafemap *bmsafemap;
 	struct jnewblk *jnewblk;
 	struct ufsmount *ump;
 	struct worklist *wk;
 	struct fs *fs;
 #ifdef SUJ_DEBUG
 	uint8_t *blksfree;
 	struct cg *cgp;
 	ufs2_daddr_t jstart;
 	ufs2_daddr_t jend;
 	ufs2_daddr_t end;
 	long bno;
 	int i;
 #endif
 
 	CTR3(KTR_SUJ,
 	    "softdep_setup_blkfree: blkno %jd frags %d wk head %p",
 	    blkno, frags, wkhd);
 
 	ump = VFSTOUFS(mp);
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ump)) != 0,
 	    ("softdep_setup_blkfree called on non-softdep filesystem"));
 	ACQUIRE_LOCK(ump);
 	/* Lookup the bmsafemap so we track when it is dirty. */
 	fs = ump->um_fs;
 	bmsafemap = bmsafemap_lookup(mp, bp, dtog(fs, blkno), NULL);
 	/*
 	 * Detach any jnewblks which have been canceled.  They must linger
 	 * until the bitmap is cleared again by ffs_blkfree() to prevent
 	 * an unjournaled allocation from hitting the disk.
 	 */
 	if (wkhd) {
 		while ((wk = LIST_FIRST(wkhd)) != NULL) {
 			CTR2(KTR_SUJ,
 			    "softdep_setup_blkfree: blkno %jd wk type %d",
 			    blkno, wk->wk_type);
 			WORKLIST_REMOVE(wk);
 			if (wk->wk_type != D_JNEWBLK) {
 				WORKLIST_INSERT(&bmsafemap->sm_freehd, wk);
 				continue;
 			}
 			jnewblk = WK_JNEWBLK(wk);
 			KASSERT(jnewblk->jn_state & GOINGAWAY,
 			    ("softdep_setup_blkfree: jnewblk not canceled."));
 #ifdef SUJ_DEBUG
 			/*
 			 * Assert that this block is free in the bitmap
 			 * before we discard the jnewblk.
 			 */
 			cgp = (struct cg *)bp->b_data;
 			blksfree = cg_blksfree(cgp);
 			bno = dtogd(fs, jnewblk->jn_blkno);
 			for (i = jnewblk->jn_oldfrags;
 			    i < jnewblk->jn_frags; i++) {
 				if (isset(blksfree, bno + i))
 					continue;
 				panic("softdep_setup_blkfree: not free");
 			}
 #endif
 			/*
 			 * Even if it's not attached we can free immediately
 			 * as the new bitmap is correct.
 			 */
 			wk->wk_state |= COMPLETE | ATTACHED;
 			free_jnewblk(jnewblk);
 		}
 	}
 
 #ifdef SUJ_DEBUG
 	/*
 	 * Assert that we are not freeing a block which has an outstanding
 	 * allocation dependency.
 	 */
 	fs = VFSTOUFS(mp)->um_fs;
 	bmsafemap = bmsafemap_lookup(mp, bp, dtog(fs, blkno), NULL);
 	end = blkno + frags;
 	LIST_FOREACH(jnewblk, &bmsafemap->sm_jnewblkhd, jn_deps) {
 		/*
 		 * Don't match against blocks that will be freed when the
 		 * background write is done.
 		 */
 		if ((jnewblk->jn_state & (ATTACHED | COMPLETE | DEPCOMPLETE)) ==
 		    (COMPLETE | DEPCOMPLETE))
 			continue;
 		jstart = jnewblk->jn_blkno + jnewblk->jn_oldfrags;
 		jend = jnewblk->jn_blkno + jnewblk->jn_frags;
 		if ((blkno >= jstart && blkno < jend) ||
 		    (end > jstart && end <= jend)) {
 			printf("state 0x%X %jd - %d %d dep %p\n",
 			    jnewblk->jn_state, jnewblk->jn_blkno,
 			    jnewblk->jn_oldfrags, jnewblk->jn_frags,
 			    jnewblk->jn_dep);
 			panic("softdep_setup_blkfree: "
 			    "%jd-%jd(%d) overlaps with %jd-%jd",
 			    blkno, end, frags, jstart, jend);
 		}
 	}
 #endif
 	FREE_LOCK(ump);
 }
 
 /*
  * Revert a block allocation when the journal record that describes it
  * is not yet written.
  */
 static int
 jnewblk_rollback(jnewblk, fs, cgp, blksfree)
 	struct jnewblk *jnewblk;
 	struct fs *fs;
 	struct cg *cgp;
 	uint8_t *blksfree;
 {
 	ufs1_daddr_t fragno;
 	long cgbno, bbase;
 	int frags, blk;
 	int i;
 
 	frags = 0;
 	cgbno = dtogd(fs, jnewblk->jn_blkno);
 	/*
 	 * We have to test which frags need to be rolled back.  We may
 	 * be operating on a stale copy when doing background writes.
 	 */
 	for (i = jnewblk->jn_oldfrags; i < jnewblk->jn_frags; i++)
 		if (isclr(blksfree, cgbno + i))
 			frags++;
 	if (frags == 0)
 		return (0);
 	/*
 	 * This is mostly ffs_blkfree() sans some validation and
 	 * superblock updates.
 	 */
 	if (frags == fs->fs_frag) {
 		fragno = fragstoblks(fs, cgbno);
 		ffs_setblock(fs, blksfree, fragno);
 		ffs_clusteracct(fs, cgp, fragno, 1);
 		cgp->cg_cs.cs_nbfree++;
 	} else {
 		cgbno += jnewblk->jn_oldfrags;
 		bbase = cgbno - fragnum(fs, cgbno);
 		/* Decrement the old frags.  */
 		blk = blkmap(fs, blksfree, bbase);
 		ffs_fragacct(fs, blk, cgp->cg_frsum, -1);
 		/* Deallocate the fragment */
 		for (i = 0; i < frags; i++)
 			setbit(blksfree, cgbno + i);
 		cgp->cg_cs.cs_nffree += frags;
 		/* Add back in counts associated with the new frags */
 		blk = blkmap(fs, blksfree, bbase);
 		ffs_fragacct(fs, blk, cgp->cg_frsum, 1);
 		/* If a complete block has been reassembled, account for it. */
 		fragno = fragstoblks(fs, bbase);
 		if (ffs_isblock(fs, blksfree, fragno)) {
 			cgp->cg_cs.cs_nffree -= fs->fs_frag;
 			ffs_clusteracct(fs, cgp, fragno, 1);
 			cgp->cg_cs.cs_nbfree++;
 		}
 	}
 	stat_jnewblk++;
 	jnewblk->jn_state &= ~ATTACHED;
 	jnewblk->jn_state |= UNDONE;
 
 	return (frags);
 }
 
 static void
 initiate_write_bmsafemap(bmsafemap, bp)
 	struct bmsafemap *bmsafemap;
 	struct buf *bp;			/* The cg block. */
 {
 	struct jaddref *jaddref;
 	struct jnewblk *jnewblk;
 	uint8_t *inosused;
 	uint8_t *blksfree;
 	struct cg *cgp;
 	struct fs *fs;
 	ino_t ino;
 
 	if (bmsafemap->sm_state & IOSTARTED)
 		return;
 	bmsafemap->sm_state |= IOSTARTED;
 	/*
 	 * Clear any inode allocations which are pending journal writes.
 	 */
 	if (LIST_FIRST(&bmsafemap->sm_jaddrefhd) != NULL) {
 		cgp = (struct cg *)bp->b_data;
 		fs = VFSTOUFS(bmsafemap->sm_list.wk_mp)->um_fs;
 		inosused = cg_inosused(cgp);
 		LIST_FOREACH(jaddref, &bmsafemap->sm_jaddrefhd, ja_bmdeps) {
 			ino = jaddref->ja_ino % fs->fs_ipg;
 			if (isset(inosused, ino)) {
 				if ((jaddref->ja_mode & IFMT) == IFDIR)
 					cgp->cg_cs.cs_ndir--;
 				cgp->cg_cs.cs_nifree++;
 				clrbit(inosused, ino);
 				jaddref->ja_state &= ~ATTACHED;
 				jaddref->ja_state |= UNDONE;
 				stat_jaddref++;
 			} else
 				panic("initiate_write_bmsafemap: inode %ju "
 				    "marked free", (uintmax_t)jaddref->ja_ino);
 		}
 	}
 	/*
 	 * Clear any block allocations which are pending journal writes.
 	 */
 	if (LIST_FIRST(&bmsafemap->sm_jnewblkhd) != NULL) {
 		cgp = (struct cg *)bp->b_data;
 		fs = VFSTOUFS(bmsafemap->sm_list.wk_mp)->um_fs;
 		blksfree = cg_blksfree(cgp);
 		LIST_FOREACH(jnewblk, &bmsafemap->sm_jnewblkhd, jn_deps) {
 			if (jnewblk_rollback(jnewblk, fs, cgp, blksfree))
 				continue;
 			panic("initiate_write_bmsafemap: block %jd "
 			    "marked free", jnewblk->jn_blkno);
 		}
 	}
 	/*
 	 * Move allocation lists to the written lists so they can be
 	 * cleared once the block write is complete.
 	 */
 	LIST_SWAP(&bmsafemap->sm_inodedephd, &bmsafemap->sm_inodedepwr,
 	    inodedep, id_deps);
 	LIST_SWAP(&bmsafemap->sm_newblkhd, &bmsafemap->sm_newblkwr,
 	    newblk, nb_deps);
 	LIST_SWAP(&bmsafemap->sm_freehd, &bmsafemap->sm_freewr, worklist,
 	    wk_list);
 }
 
 /*
  * This routine is called during the completion interrupt
  * service routine for a disk write (from the procedure called
  * by the device driver to inform the filesystem caches of
  * a request completion).  It should be called early in this
  * procedure, before the block is made available to other
  * processes or other routines are called.
  *
  */
 static void 
 softdep_disk_write_complete(bp)
 	struct buf *bp;		/* describes the completed disk write */
 {
 	struct worklist *wk;
 	struct worklist *owk;
 	struct ufsmount *ump;
 	struct workhead reattach;
 	struct freeblks *freeblks;
 	struct buf *sbp;
 
 	/*
 	 * If an error occurred while doing the write, then the data
 	 * has not hit the disk and the dependencies cannot be unrolled.
 	 */
 	if ((bp->b_ioflags & BIO_ERROR) != 0 && (bp->b_flags & B_INVAL) == 0)
 		return;
 	if ((wk = LIST_FIRST(&bp->b_dep)) == NULL)
 		return;
 	ump = VFSTOUFS(wk->wk_mp);
 	LIST_INIT(&reattach);
 	/*
 	 * This lock must not be released anywhere in this code segment.
 	 */
 	sbp = NULL;
 	owk = NULL;
 	ACQUIRE_LOCK(ump);
 	while ((wk = LIST_FIRST(&bp->b_dep)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		atomic_add_long(&dep_write[wk->wk_type], 1);
 		if (wk == owk)
 			panic("duplicate worklist: %p\n", wk);
 		owk = wk;
 		switch (wk->wk_type) {
 
 		case D_PAGEDEP:
 			if (handle_written_filepage(WK_PAGEDEP(wk), bp))
 				WORKLIST_INSERT(&reattach, wk);
 			continue;
 
 		case D_INODEDEP:
 			if (handle_written_inodeblock(WK_INODEDEP(wk), bp))
 				WORKLIST_INSERT(&reattach, wk);
 			continue;
 
 		case D_BMSAFEMAP:
 			if (handle_written_bmsafemap(WK_BMSAFEMAP(wk), bp))
 				WORKLIST_INSERT(&reattach, wk);
 			continue;
 
 		case D_MKDIR:
 			handle_written_mkdir(WK_MKDIR(wk), MKDIR_BODY);
 			continue;
 
 		case D_ALLOCDIRECT:
 			wk->wk_state |= COMPLETE;
 			handle_allocdirect_partdone(WK_ALLOCDIRECT(wk), NULL);
 			continue;
 
 		case D_ALLOCINDIR:
 			wk->wk_state |= COMPLETE;
 			handle_allocindir_partdone(WK_ALLOCINDIR(wk));
 			continue;
 
 		case D_INDIRDEP:
 			if (handle_written_indirdep(WK_INDIRDEP(wk), bp, &sbp))
 				WORKLIST_INSERT(&reattach, wk);
 			continue;
 
 		case D_FREEBLKS:
 			wk->wk_state |= COMPLETE;
 			freeblks = WK_FREEBLKS(wk);
 			if ((wk->wk_state & ALLCOMPLETE) == ALLCOMPLETE &&
 			    LIST_EMPTY(&freeblks->fb_jblkdephd))
 				add_to_worklist(wk, WK_NODELAY);
 			continue;
 
 		case D_FREEWORK:
 			handle_written_freework(WK_FREEWORK(wk));
 			break;
 
 		case D_JSEGDEP:
 			free_jsegdep(WK_JSEGDEP(wk));
 			continue;
 
 		case D_JSEG:
 			handle_written_jseg(WK_JSEG(wk), bp);
 			continue;
 
 		case D_SBDEP:
 			if (handle_written_sbdep(WK_SBDEP(wk), bp))
 				WORKLIST_INSERT(&reattach, wk);
 			continue;
 
 		case D_FREEDEP:
 			free_freedep(WK_FREEDEP(wk));
 			continue;
 
 		default:
 			panic("handle_disk_write_complete: Unknown type %s",
 			    TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 	}
 	/*
 	 * Reattach any requests that must be redone.
 	 */
 	while ((wk = LIST_FIRST(&reattach)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		WORKLIST_INSERT(&bp->b_dep, wk);
 	}
 	FREE_LOCK(ump);
 	if (sbp)
 		brelse(sbp);
 }
 
 /*
  * Called from within softdep_disk_write_complete above. Note that
  * this routine is always called from interrupt level with further
  * splbio interrupts blocked.
  */
 static void 
 handle_allocdirect_partdone(adp, wkhd)
 	struct allocdirect *adp;	/* the completed allocdirect */
 	struct workhead *wkhd;		/* Work to do when inode is writtne. */
 {
 	struct allocdirectlst *listhead;
 	struct allocdirect *listadp;
 	struct inodedep *inodedep;
 	long bsize;
 
 	if ((adp->ad_state & ALLCOMPLETE) != ALLCOMPLETE)
 		return;
 	/*
 	 * The on-disk inode cannot claim to be any larger than the last
 	 * fragment that has been written. Otherwise, the on-disk inode
 	 * might have fragments that were not the last block in the file
 	 * which would corrupt the filesystem. Thus, we cannot free any
 	 * allocdirects after one whose ad_oldblkno claims a fragment as
 	 * these blocks must be rolled back to zero before writing the inode.
 	 * We check the currently active set of allocdirects in id_inoupdt
 	 * or id_extupdt as appropriate.
 	 */
 	inodedep = adp->ad_inodedep;
 	bsize = inodedep->id_fs->fs_bsize;
 	if (adp->ad_state & EXTDATA)
 		listhead = &inodedep->id_extupdt;
 	else
 		listhead = &inodedep->id_inoupdt;
 	TAILQ_FOREACH(listadp, listhead, ad_next) {
 		/* found our block */
 		if (listadp == adp)
 			break;
 		/* continue if ad_oldlbn is not a fragment */
 		if (listadp->ad_oldsize == 0 ||
 		    listadp->ad_oldsize == bsize)
 			continue;
 		/* hit a fragment */
 		return;
 	}
 	/*
 	 * If we have reached the end of the current list without
 	 * finding the just finished dependency, then it must be
 	 * on the future dependency list. Future dependencies cannot
 	 * be freed until they are moved to the current list.
 	 */
 	if (listadp == NULL) {
 #ifdef DEBUG
 		if (adp->ad_state & EXTDATA)
 			listhead = &inodedep->id_newextupdt;
 		else
 			listhead = &inodedep->id_newinoupdt;
 		TAILQ_FOREACH(listadp, listhead, ad_next)
 			/* found our block */
 			if (listadp == adp)
 				break;
 		if (listadp == NULL)
 			panic("handle_allocdirect_partdone: lost dep");
 #endif /* DEBUG */
 		return;
 	}
 	/*
 	 * If we have found the just finished dependency, then queue
 	 * it along with anything that follows it that is complete.
 	 * Since the pointer has not yet been written in the inode
 	 * as the dependency prevents it, place the allocdirect on the
 	 * bufwait list where it will be freed once the pointer is
 	 * valid.
 	 */
 	if (wkhd == NULL)
 		wkhd = &inodedep->id_bufwait;
 	for (; adp; adp = listadp) {
 		listadp = TAILQ_NEXT(adp, ad_next);
 		if ((adp->ad_state & ALLCOMPLETE) != ALLCOMPLETE)
 			return;
 		TAILQ_REMOVE(listhead, adp, ad_next);
 		WORKLIST_INSERT(wkhd, &adp->ad_block.nb_list);
 	}
 }
 
 /*
  * Called from within softdep_disk_write_complete above.  This routine
  * completes successfully written allocindirs.
  */
 static void
 handle_allocindir_partdone(aip)
 	struct allocindir *aip;		/* the completed allocindir */
 {
 	struct indirdep *indirdep;
 
 	if ((aip->ai_state & ALLCOMPLETE) != ALLCOMPLETE)
 		return;
 	indirdep = aip->ai_indirdep;
 	LIST_REMOVE(aip, ai_next);
 	/*
 	 * Don't set a pointer while the buffer is undergoing IO or while
 	 * we have active truncations.
 	 */
 	if (indirdep->ir_state & UNDONE || !TAILQ_EMPTY(&indirdep->ir_trunc)) {
 		LIST_INSERT_HEAD(&indirdep->ir_donehd, aip, ai_next);
 		return;
 	}
 	if (indirdep->ir_state & UFS1FMT)
 		((ufs1_daddr_t *)indirdep->ir_savebp->b_data)[aip->ai_offset] =
 		    aip->ai_newblkno;
 	else
 		((ufs2_daddr_t *)indirdep->ir_savebp->b_data)[aip->ai_offset] =
 		    aip->ai_newblkno;
 	/*
 	 * Await the pointer write before freeing the allocindir.
 	 */
 	LIST_INSERT_HEAD(&indirdep->ir_writehd, aip, ai_next);
 }
 
 /*
  * Release segments held on a jwork list.
  */
 static void
 handle_jwork(wkhd)
 	struct workhead *wkhd;
 {
 	struct worklist *wk;
 
 	while ((wk = LIST_FIRST(wkhd)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		switch (wk->wk_type) {
 		case D_JSEGDEP:
 			free_jsegdep(WK_JSEGDEP(wk));
 			continue;
 		case D_FREEDEP:
 			free_freedep(WK_FREEDEP(wk));
 			continue;
 		case D_FREEFRAG:
 			rele_jseg(WK_JSEG(WK_FREEFRAG(wk)->ff_jdep));
 			WORKITEM_FREE(wk, D_FREEFRAG);
 			continue;
 		case D_FREEWORK:
 			handle_written_freework(WK_FREEWORK(wk));
 			continue;
 		default:
 			panic("handle_jwork: Unknown type %s\n",
 			    TYPENAME(wk->wk_type));
 		}
 	}
 }
 
 /*
  * Handle the bufwait list on an inode when it is safe to release items
  * held there.  This normally happens after an inode block is written but
  * may be delayed and handled later if there are pending journal items that
  * are not yet safe to be released.
  */
 static struct freefile *
 handle_bufwait(inodedep, refhd)
 	struct inodedep *inodedep;
 	struct workhead *refhd;
 {
 	struct jaddref *jaddref;
 	struct freefile *freefile;
 	struct worklist *wk;
 
 	freefile = NULL;
 	while ((wk = LIST_FIRST(&inodedep->id_bufwait)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		switch (wk->wk_type) {
 		case D_FREEFILE:
 			/*
 			 * We defer adding freefile to the worklist
 			 * until all other additions have been made to
 			 * ensure that it will be done after all the
 			 * old blocks have been freed.
 			 */
 			if (freefile != NULL)
 				panic("handle_bufwait: freefile");
 			freefile = WK_FREEFILE(wk);
 			continue;
 
 		case D_MKDIR:
 			handle_written_mkdir(WK_MKDIR(wk), MKDIR_PARENT);
 			continue;
 
 		case D_DIRADD:
 			diradd_inode_written(WK_DIRADD(wk), inodedep);
 			continue;
 
 		case D_FREEFRAG:
 			wk->wk_state |= COMPLETE;
 			if ((wk->wk_state & ALLCOMPLETE) == ALLCOMPLETE)
 				add_to_worklist(wk, 0);
 			continue;
 
 		case D_DIRREM:
 			wk->wk_state |= COMPLETE;
 			add_to_worklist(wk, 0);
 			continue;
 
 		case D_ALLOCDIRECT:
 		case D_ALLOCINDIR:
 			free_newblk(WK_NEWBLK(wk));
 			continue;
 
 		case D_JNEWBLK:
 			wk->wk_state |= COMPLETE;
 			free_jnewblk(WK_JNEWBLK(wk));
 			continue;
 
 		/*
 		 * Save freed journal segments and add references on
 		 * the supplied list which will delay their release
 		 * until the cg bitmap is cleared on disk.
 		 */
 		case D_JSEGDEP:
 			if (refhd == NULL)
 				free_jsegdep(WK_JSEGDEP(wk));
 			else
 				WORKLIST_INSERT(refhd, wk);
 			continue;
 
 		case D_JADDREF:
 			jaddref = WK_JADDREF(wk);
 			TAILQ_REMOVE(&inodedep->id_inoreflst, &jaddref->ja_ref,
 			    if_deps);
 			/*
 			 * Transfer any jaddrefs to the list to be freed with
 			 * the bitmap if we're handling a removed file.
 			 */
 			if (refhd == NULL) {
 				wk->wk_state |= COMPLETE;
 				free_jaddref(jaddref);
 			} else
 				WORKLIST_INSERT(refhd, wk);
 			continue;
 
 		default:
 			panic("handle_bufwait: Unknown type %p(%s)",
 			    wk, TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 	}
 	return (freefile);
 }
 /*
  * Called from within softdep_disk_write_complete above to restore
  * in-memory inode block contents to their most up-to-date state. Note
  * that this routine is always called from interrupt level with further
  * splbio interrupts blocked.
  */
 static int 
 handle_written_inodeblock(inodedep, bp)
 	struct inodedep *inodedep;
 	struct buf *bp;		/* buffer containing the inode block */
 {
 	struct freefile *freefile;
 	struct allocdirect *adp, *nextadp;
 	struct ufs1_dinode *dp1 = NULL;
 	struct ufs2_dinode *dp2 = NULL;
 	struct workhead wkhd;
 	int hadchanges, fstype;
 	ino_t freelink;
 
 	LIST_INIT(&wkhd);
 	hadchanges = 0;
 	freefile = NULL;
 	if ((inodedep->id_state & IOSTARTED) == 0)
 		panic("handle_written_inodeblock: not started");
 	inodedep->id_state &= ~IOSTARTED;
 	if (inodedep->id_fs->fs_magic == FS_UFS1_MAGIC) {
 		fstype = UFS1;
 		dp1 = (struct ufs1_dinode *)bp->b_data +
 		    ino_to_fsbo(inodedep->id_fs, inodedep->id_ino);
 		freelink = dp1->di_freelink;
 	} else {
 		fstype = UFS2;
 		dp2 = (struct ufs2_dinode *)bp->b_data +
 		    ino_to_fsbo(inodedep->id_fs, inodedep->id_ino);
 		freelink = dp2->di_freelink;
 	}
 	/*
 	 * Leave this inodeblock dirty until it's in the list.
 	 */
 	if ((inodedep->id_state & (UNLINKED | UNLINKONLIST)) == UNLINKED) {
 		struct inodedep *inon;
 
 		inon = TAILQ_NEXT(inodedep, id_unlinked);
 		if ((inon == NULL && freelink == 0) ||
 		    (inon && inon->id_ino == freelink)) {
 			if (inon)
 				inon->id_state |= UNLINKPREV;
 			inodedep->id_state |= UNLINKNEXT;
 		}
 		hadchanges = 1;
 	}
 	/*
 	 * If we had to rollback the inode allocation because of
 	 * bitmaps being incomplete, then simply restore it.
 	 * Keep the block dirty so that it will not be reclaimed until
 	 * all associated dependencies have been cleared and the
 	 * corresponding updates written to disk.
 	 */
 	if (inodedep->id_savedino1 != NULL) {
 		hadchanges = 1;
 		if (fstype == UFS1)
 			*dp1 = *inodedep->id_savedino1;
 		else
 			*dp2 = *inodedep->id_savedino2;
 		free(inodedep->id_savedino1, M_SAVEDINO);
 		inodedep->id_savedino1 = NULL;
 		if ((bp->b_flags & B_DELWRI) == 0)
 			stat_inode_bitmap++;
 		bdirty(bp);
 		/*
 		 * If the inode is clear here and GOINGAWAY it will never
 		 * be written.  Process the bufwait and clear any pending
 		 * work which may include the freefile.
 		 */
 		if (inodedep->id_state & GOINGAWAY)
 			goto bufwait;
 		return (1);
 	}
 	inodedep->id_state |= COMPLETE;
 	/*
 	 * Roll forward anything that had to be rolled back before 
 	 * the inode could be updated.
 	 */
 	for (adp = TAILQ_FIRST(&inodedep->id_inoupdt); adp; adp = nextadp) {
 		nextadp = TAILQ_NEXT(adp, ad_next);
 		if (adp->ad_state & ATTACHED)
 			panic("handle_written_inodeblock: new entry");
 		if (fstype == UFS1) {
 			if (adp->ad_offset < NDADDR) {
 				if (dp1->di_db[adp->ad_offset]!=adp->ad_oldblkno)
 					panic("%s %s #%jd mismatch %d != %jd",
 					    "handle_written_inodeblock:",
 					    "direct pointer",
 					    (intmax_t)adp->ad_offset,
 					    dp1->di_db[adp->ad_offset],
 					    (intmax_t)adp->ad_oldblkno);
 				dp1->di_db[adp->ad_offset] = adp->ad_newblkno;
 			} else {
 				if (dp1->di_ib[adp->ad_offset - NDADDR] != 0)
 					panic("%s: %s #%jd allocated as %d",
 					    "handle_written_inodeblock",
 					    "indirect pointer",
 					    (intmax_t)adp->ad_offset - NDADDR,
 					    dp1->di_ib[adp->ad_offset - NDADDR]);
 				dp1->di_ib[adp->ad_offset - NDADDR] =
 				    adp->ad_newblkno;
 			}
 		} else {
 			if (adp->ad_offset < NDADDR) {
 				if (dp2->di_db[adp->ad_offset]!=adp->ad_oldblkno)
 					panic("%s: %s #%jd %s %jd != %jd",
 					    "handle_written_inodeblock",
 					    "direct pointer",
 					    (intmax_t)adp->ad_offset, "mismatch",
 					    (intmax_t)dp2->di_db[adp->ad_offset],
 					    (intmax_t)adp->ad_oldblkno);
 				dp2->di_db[adp->ad_offset] = adp->ad_newblkno;
 			} else {
 				if (dp2->di_ib[adp->ad_offset - NDADDR] != 0)
 					panic("%s: %s #%jd allocated as %jd",
 					    "handle_written_inodeblock",
 					    "indirect pointer",
 					    (intmax_t)adp->ad_offset - NDADDR,
 					    (intmax_t)
 					    dp2->di_ib[adp->ad_offset - NDADDR]);
 				dp2->di_ib[adp->ad_offset - NDADDR] =
 				    adp->ad_newblkno;
 			}
 		}
 		adp->ad_state &= ~UNDONE;
 		adp->ad_state |= ATTACHED;
 		hadchanges = 1;
 	}
 	for (adp = TAILQ_FIRST(&inodedep->id_extupdt); adp; adp = nextadp) {
 		nextadp = TAILQ_NEXT(adp, ad_next);
 		if (adp->ad_state & ATTACHED)
 			panic("handle_written_inodeblock: new entry");
 		if (dp2->di_extb[adp->ad_offset] != adp->ad_oldblkno)
 			panic("%s: direct pointers #%jd %s %jd != %jd",
 			    "handle_written_inodeblock",
 			    (intmax_t)adp->ad_offset, "mismatch",
 			    (intmax_t)dp2->di_extb[adp->ad_offset],
 			    (intmax_t)adp->ad_oldblkno);
 		dp2->di_extb[adp->ad_offset] = adp->ad_newblkno;
 		adp->ad_state &= ~UNDONE;
 		adp->ad_state |= ATTACHED;
 		hadchanges = 1;
 	}
 	if (hadchanges && (bp->b_flags & B_DELWRI) == 0)
 		stat_direct_blk_ptrs++;
 	/*
 	 * Reset the file size to its most up-to-date value.
 	 */
 	if (inodedep->id_savedsize == -1 || inodedep->id_savedextsize == -1)
 		panic("handle_written_inodeblock: bad size");
 	if (inodedep->id_savednlink > LINK_MAX)
 		panic("handle_written_inodeblock: Invalid link count "
 		    "%d for inodedep %p", inodedep->id_savednlink, inodedep);
 	if (fstype == UFS1) {
 		if (dp1->di_nlink != inodedep->id_savednlink) { 
 			dp1->di_nlink = inodedep->id_savednlink;
 			hadchanges = 1;
 		}
 		if (dp1->di_size != inodedep->id_savedsize) {
 			dp1->di_size = inodedep->id_savedsize;
 			hadchanges = 1;
 		}
 	} else {
 		if (dp2->di_nlink != inodedep->id_savednlink) { 
 			dp2->di_nlink = inodedep->id_savednlink;
 			hadchanges = 1;
 		}
 		if (dp2->di_size != inodedep->id_savedsize) {
 			dp2->di_size = inodedep->id_savedsize;
 			hadchanges = 1;
 		}
 		if (dp2->di_extsize != inodedep->id_savedextsize) {
 			dp2->di_extsize = inodedep->id_savedextsize;
 			hadchanges = 1;
 		}
 	}
 	inodedep->id_savedsize = -1;
 	inodedep->id_savedextsize = -1;
 	inodedep->id_savednlink = -1;
 	/*
 	 * If there were any rollbacks in the inode block, then it must be
 	 * marked dirty so that its will eventually get written back in
 	 * its correct form.
 	 */
 	if (hadchanges)
 		bdirty(bp);
 bufwait:
 	/*
 	 * Process any allocdirects that completed during the update.
 	 */
 	if ((adp = TAILQ_FIRST(&inodedep->id_inoupdt)) != NULL)
 		handle_allocdirect_partdone(adp, &wkhd);
 	if ((adp = TAILQ_FIRST(&inodedep->id_extupdt)) != NULL)
 		handle_allocdirect_partdone(adp, &wkhd);
 	/*
 	 * Process deallocations that were held pending until the
 	 * inode had been written to disk. Freeing of the inode
 	 * is delayed until after all blocks have been freed to
 	 * avoid creation of new <vfsid, inum, lbn> triples
 	 * before the old ones have been deleted.  Completely
 	 * unlinked inodes are not processed until the unlinked
 	 * inode list is written or the last reference is removed.
 	 */
 	if ((inodedep->id_state & (UNLINKED | UNLINKONLIST)) != UNLINKED) {
 		freefile = handle_bufwait(inodedep, NULL);
 		if (freefile && !LIST_EMPTY(&wkhd)) {
 			WORKLIST_INSERT(&wkhd, &freefile->fx_list);
 			freefile = NULL;
 		}
 	}
 	/*
 	 * Move rolled forward dependency completions to the bufwait list
 	 * now that those that were already written have been processed.
 	 */
 	if (!LIST_EMPTY(&wkhd) && hadchanges == 0)
 		panic("handle_written_inodeblock: bufwait but no changes");
 	jwork_move(&inodedep->id_bufwait, &wkhd);
 
 	if (freefile != NULL) {
 		/*
 		 * If the inode is goingaway it was never written.  Fake up
 		 * the state here so free_inodedep() can succeed.
 		 */
 		if (inodedep->id_state & GOINGAWAY)
 			inodedep->id_state |= COMPLETE | DEPCOMPLETE;
 		if (free_inodedep(inodedep) == 0)
 			panic("handle_written_inodeblock: live inodedep %p",
 			    inodedep);
 		add_to_worklist(&freefile->fx_list, 0);
 		return (0);
 	}
 
 	/*
 	 * If no outstanding dependencies, free it.
 	 */
 	if (free_inodedep(inodedep) ||
 	    (TAILQ_FIRST(&inodedep->id_inoreflst) == 0 &&
 	     TAILQ_FIRST(&inodedep->id_inoupdt) == 0 &&
 	     TAILQ_FIRST(&inodedep->id_extupdt) == 0 &&
 	     LIST_FIRST(&inodedep->id_bufwait) == 0))
 		return (0);
 	return (hadchanges);
 }
 
 static int
 handle_written_indirdep(indirdep, bp, bpp)
 	struct indirdep *indirdep;
 	struct buf *bp;
 	struct buf **bpp;
 {
 	struct allocindir *aip;
 	struct buf *sbp;
 	int chgs;
 
 	if (indirdep->ir_state & GOINGAWAY)
 		panic("handle_written_indirdep: indirdep gone");
 	if ((indirdep->ir_state & IOSTARTED) == 0)
 		panic("handle_written_indirdep: IO not started");
 	chgs = 0;
 	/*
 	 * If there were rollbacks revert them here.
 	 */
 	if (indirdep->ir_saveddata) {
 		bcopy(indirdep->ir_saveddata, bp->b_data, bp->b_bcount);
 		if (TAILQ_EMPTY(&indirdep->ir_trunc)) {
 			free(indirdep->ir_saveddata, M_INDIRDEP);
 			indirdep->ir_saveddata = NULL;
 		}
 		chgs = 1;
 	}
 	indirdep->ir_state &= ~(UNDONE | IOSTARTED);
 	indirdep->ir_state |= ATTACHED;
 	/*
 	 * Move allocindirs with written pointers to the completehd if
 	 * the indirdep's pointer is not yet written.  Otherwise
 	 * free them here.
 	 */
 	while ((aip = LIST_FIRST(&indirdep->ir_writehd)) != 0) {
 		LIST_REMOVE(aip, ai_next);
 		if ((indirdep->ir_state & DEPCOMPLETE) == 0) {
 			LIST_INSERT_HEAD(&indirdep->ir_completehd, aip,
 			    ai_next);
 			newblk_freefrag(&aip->ai_block);
 			continue;
 		}
 		free_newblk(&aip->ai_block);
 	}
 	/*
 	 * Move allocindirs that have finished dependency processing from
 	 * the done list to the write list after updating the pointers.
 	 */
 	if (TAILQ_EMPTY(&indirdep->ir_trunc)) {
 		while ((aip = LIST_FIRST(&indirdep->ir_donehd)) != 0) {
 			handle_allocindir_partdone(aip);
 			if (aip == LIST_FIRST(&indirdep->ir_donehd))
 				panic("disk_write_complete: not gone");
 			chgs = 1;
 		}
 	}
 	/*
 	 * Preserve the indirdep if there were any changes or if it is not
 	 * yet valid on disk.
 	 */
 	if (chgs) {
 		stat_indir_blk_ptrs++;
 		bdirty(bp);
 		return (1);
 	}
 	/*
 	 * If there were no changes we can discard the savedbp and detach
 	 * ourselves from the buf.  We are only carrying completed pointers
 	 * in this case.
 	 */
 	sbp = indirdep->ir_savebp;
 	sbp->b_flags |= B_INVAL | B_NOCACHE;
 	indirdep->ir_savebp = NULL;
 	indirdep->ir_bp = NULL;
 	if (*bpp != NULL)
 		panic("handle_written_indirdep: bp already exists.");
 	*bpp = sbp;
 	/*
 	 * The indirdep may not be freed until its parent points at it.
 	 */
 	if (indirdep->ir_state & DEPCOMPLETE)
 		free_indirdep(indirdep);
 
 	return (0);
 }
 
 /*
  * Process a diradd entry after its dependent inode has been written.
  * This routine must be called with splbio interrupts blocked.
  */
 static void
 diradd_inode_written(dap, inodedep)
 	struct diradd *dap;
 	struct inodedep *inodedep;
 {
 
 	dap->da_state |= COMPLETE;
 	complete_diradd(dap);
 	WORKLIST_INSERT(&inodedep->id_pendinghd, &dap->da_list);
 }
 
 /*
  * Returns true if the bmsafemap will have rollbacks when written.  Must only
  * be called with the per-filesystem lock and the buf lock on the cg held.
  */
 static int
 bmsafemap_backgroundwrite(bmsafemap, bp)
 	struct bmsafemap *bmsafemap;
 	struct buf *bp;
 {
 	int dirty;
 
 	LOCK_OWNED(VFSTOUFS(bmsafemap->sm_list.wk_mp));
 	dirty = !LIST_EMPTY(&bmsafemap->sm_jaddrefhd) | 
 	    !LIST_EMPTY(&bmsafemap->sm_jnewblkhd);
 	/*
 	 * If we're initiating a background write we need to process the
 	 * rollbacks as they exist now, not as they exist when IO starts.
 	 * No other consumers will look at the contents of the shadowed
 	 * buf so this is safe to do here.
 	 */
 	if (bp->b_xflags & BX_BKGRDMARKER)
 		initiate_write_bmsafemap(bmsafemap, bp);
 
 	return (dirty);
 }
 
 /*
  * Re-apply an allocation when a cg write is complete.
  */
 static int
 jnewblk_rollforward(jnewblk, fs, cgp, blksfree)
 	struct jnewblk *jnewblk;
 	struct fs *fs;
 	struct cg *cgp;
 	uint8_t *blksfree;
 {
 	ufs1_daddr_t fragno;
 	ufs2_daddr_t blkno;
 	long cgbno, bbase;
 	int frags, blk;
 	int i;
 
 	frags = 0;
 	cgbno = dtogd(fs, jnewblk->jn_blkno);
 	for (i = jnewblk->jn_oldfrags; i < jnewblk->jn_frags; i++) {
 		if (isclr(blksfree, cgbno + i))
 			panic("jnewblk_rollforward: re-allocated fragment");
 		frags++;
 	}
 	if (frags == fs->fs_frag) {
 		blkno = fragstoblks(fs, cgbno);
 		ffs_clrblock(fs, blksfree, (long)blkno);
 		ffs_clusteracct(fs, cgp, blkno, -1);
 		cgp->cg_cs.cs_nbfree--;
 	} else {
 		bbase = cgbno - fragnum(fs, cgbno);
 		cgbno += jnewblk->jn_oldfrags;
                 /* If a complete block had been reassembled, account for it. */
 		fragno = fragstoblks(fs, bbase);
 		if (ffs_isblock(fs, blksfree, fragno)) {
 			cgp->cg_cs.cs_nffree += fs->fs_frag;
 			ffs_clusteracct(fs, cgp, fragno, -1);
 			cgp->cg_cs.cs_nbfree--;
 		}
 		/* Decrement the old frags.  */
 		blk = blkmap(fs, blksfree, bbase);
 		ffs_fragacct(fs, blk, cgp->cg_frsum, -1);
 		/* Allocate the fragment */
 		for (i = 0; i < frags; i++)
 			clrbit(blksfree, cgbno + i);
 		cgp->cg_cs.cs_nffree -= frags;
 		/* Add back in counts associated with the new frags */
 		blk = blkmap(fs, blksfree, bbase);
 		ffs_fragacct(fs, blk, cgp->cg_frsum, 1);
 	}
 	return (frags);
 }
 
 /*
  * Complete a write to a bmsafemap structure.  Roll forward any bitmap
  * changes if it's not a background write.  Set all written dependencies 
  * to DEPCOMPLETE and free the structure if possible.
  */
 static int
 handle_written_bmsafemap(bmsafemap, bp)
 	struct bmsafemap *bmsafemap;
 	struct buf *bp;
 {
 	struct newblk *newblk;
 	struct inodedep *inodedep;
 	struct jaddref *jaddref, *jatmp;
 	struct jnewblk *jnewblk, *jntmp;
 	struct ufsmount *ump;
 	uint8_t *inosused;
 	uint8_t *blksfree;
 	struct cg *cgp;
 	struct fs *fs;
 	ino_t ino;
 	int foreground;
 	int chgs;
 
 	if ((bmsafemap->sm_state & IOSTARTED) == 0)
 		panic("initiate_write_bmsafemap: Not started\n");
 	ump = VFSTOUFS(bmsafemap->sm_list.wk_mp);
 	chgs = 0;
 	bmsafemap->sm_state &= ~IOSTARTED;
 	foreground = (bp->b_xflags & BX_BKGRDMARKER) == 0;
 	/*
 	 * Release journal work that was waiting on the write.
 	 */
 	handle_jwork(&bmsafemap->sm_freewr);
 
 	/*
 	 * Restore unwritten inode allocation pending jaddref writes.
 	 */
 	if (!LIST_EMPTY(&bmsafemap->sm_jaddrefhd)) {
 		cgp = (struct cg *)bp->b_data;
 		fs = VFSTOUFS(bmsafemap->sm_list.wk_mp)->um_fs;
 		inosused = cg_inosused(cgp);
 		LIST_FOREACH_SAFE(jaddref, &bmsafemap->sm_jaddrefhd,
 		    ja_bmdeps, jatmp) {
 			if ((jaddref->ja_state & UNDONE) == 0)
 				continue;
 			ino = jaddref->ja_ino % fs->fs_ipg;
 			if (isset(inosused, ino))
 				panic("handle_written_bmsafemap: "
 				    "re-allocated inode");
 			/* Do the roll-forward only if it's a real copy. */
 			if (foreground) {
 				if ((jaddref->ja_mode & IFMT) == IFDIR)
 					cgp->cg_cs.cs_ndir++;
 				cgp->cg_cs.cs_nifree--;
 				setbit(inosused, ino);
 				chgs = 1;
 			}
 			jaddref->ja_state &= ~UNDONE;
 			jaddref->ja_state |= ATTACHED;
 			free_jaddref(jaddref);
 		}
 	}
 	/*
 	 * Restore any block allocations which are pending journal writes.
 	 */
 	if (LIST_FIRST(&bmsafemap->sm_jnewblkhd) != NULL) {
 		cgp = (struct cg *)bp->b_data;
 		fs = VFSTOUFS(bmsafemap->sm_list.wk_mp)->um_fs;
 		blksfree = cg_blksfree(cgp);
 		LIST_FOREACH_SAFE(jnewblk, &bmsafemap->sm_jnewblkhd, jn_deps,
 		    jntmp) {
 			if ((jnewblk->jn_state & UNDONE) == 0)
 				continue;
 			/* Do the roll-forward only if it's a real copy. */
 			if (foreground &&
 			    jnewblk_rollforward(jnewblk, fs, cgp, blksfree))
 				chgs = 1;
 			jnewblk->jn_state &= ~(UNDONE | NEWBLOCK);
 			jnewblk->jn_state |= ATTACHED;
 			free_jnewblk(jnewblk);
 		}
 	}
 	while ((newblk = LIST_FIRST(&bmsafemap->sm_newblkwr))) {
 		newblk->nb_state |= DEPCOMPLETE;
 		newblk->nb_state &= ~ONDEPLIST;
 		newblk->nb_bmsafemap = NULL;
 		LIST_REMOVE(newblk, nb_deps);
 		if (newblk->nb_list.wk_type == D_ALLOCDIRECT)
 			handle_allocdirect_partdone(
 			    WK_ALLOCDIRECT(&newblk->nb_list), NULL);
 		else if (newblk->nb_list.wk_type == D_ALLOCINDIR)
 			handle_allocindir_partdone(
 			    WK_ALLOCINDIR(&newblk->nb_list));
 		else if (newblk->nb_list.wk_type != D_NEWBLK)
 			panic("handle_written_bmsafemap: Unexpected type: %s",
 			    TYPENAME(newblk->nb_list.wk_type));
 	}
 	while ((inodedep = LIST_FIRST(&bmsafemap->sm_inodedepwr)) != NULL) {
 		inodedep->id_state |= DEPCOMPLETE;
 		inodedep->id_state &= ~ONDEPLIST;
 		LIST_REMOVE(inodedep, id_deps);
 		inodedep->id_bmsafemap = NULL;
 	}
 	LIST_REMOVE(bmsafemap, sm_next);
 	if (chgs == 0 && LIST_EMPTY(&bmsafemap->sm_jaddrefhd) &&
 	    LIST_EMPTY(&bmsafemap->sm_jnewblkhd) &&
 	    LIST_EMPTY(&bmsafemap->sm_newblkhd) &&
 	    LIST_EMPTY(&bmsafemap->sm_inodedephd) &&
 	    LIST_EMPTY(&bmsafemap->sm_freehd)) {
 		LIST_REMOVE(bmsafemap, sm_hash);
 		WORKITEM_FREE(bmsafemap, D_BMSAFEMAP);
 		return (0);
 	}
 	LIST_INSERT_HEAD(&ump->softdep_dirtycg, bmsafemap, sm_next);
 	if (foreground)
 		bdirty(bp);
 	return (1);
 }
 
 /*
  * Try to free a mkdir dependency.
  */
 static void
 complete_mkdir(mkdir)
 	struct mkdir *mkdir;
 {
 	struct diradd *dap;
 
 	if ((mkdir->md_state & ALLCOMPLETE) != ALLCOMPLETE)
 		return;
 	LIST_REMOVE(mkdir, md_mkdirs);
 	dap = mkdir->md_diradd;
 	dap->da_state &= ~(mkdir->md_state & (MKDIR_PARENT | MKDIR_BODY));
 	if ((dap->da_state & (MKDIR_PARENT | MKDIR_BODY)) == 0) {
 		dap->da_state |= DEPCOMPLETE;
 		complete_diradd(dap);
 	}
 	WORKITEM_FREE(mkdir, D_MKDIR);
 }
 
 /*
  * Handle the completion of a mkdir dependency.
  */
 static void
 handle_written_mkdir(mkdir, type)
 	struct mkdir *mkdir;
 	int type;
 {
 
 	if ((mkdir->md_state & (MKDIR_PARENT | MKDIR_BODY)) != type)
 		panic("handle_written_mkdir: bad type");
 	mkdir->md_state |= COMPLETE;
 	complete_mkdir(mkdir);
 }
 
 static int
 free_pagedep(pagedep)
 	struct pagedep *pagedep;
 {
 	int i;
 
 	if (pagedep->pd_state & NEWBLOCK)
 		return (0);
 	if (!LIST_EMPTY(&pagedep->pd_dirremhd))
 		return (0);
 	for (i = 0; i < DAHASHSZ; i++)
 		if (!LIST_EMPTY(&pagedep->pd_diraddhd[i]))
 			return (0);
 	if (!LIST_EMPTY(&pagedep->pd_pendinghd))
 		return (0);
 	if (!LIST_EMPTY(&pagedep->pd_jmvrefhd))
 		return (0);
 	if (pagedep->pd_state & ONWORKLIST)
 		WORKLIST_REMOVE(&pagedep->pd_list);
 	LIST_REMOVE(pagedep, pd_hash);
 	WORKITEM_FREE(pagedep, D_PAGEDEP);
 
 	return (1);
 }
 
 /*
  * Called from within softdep_disk_write_complete above.
  * A write operation was just completed. Removed inodes can
  * now be freed and associated block pointers may be committed.
  * Note that this routine is always called from interrupt level
  * with further splbio interrupts blocked.
  */
 static int 
 handle_written_filepage(pagedep, bp)
 	struct pagedep *pagedep;
 	struct buf *bp;		/* buffer containing the written page */
 {
 	struct dirrem *dirrem;
 	struct diradd *dap, *nextdap;
 	struct direct *ep;
 	int i, chgs;
 
 	if ((pagedep->pd_state & IOSTARTED) == 0)
 		panic("handle_written_filepage: not started");
 	pagedep->pd_state &= ~IOSTARTED;
 	/*
 	 * Process any directory removals that have been committed.
 	 */
 	while ((dirrem = LIST_FIRST(&pagedep->pd_dirremhd)) != NULL) {
 		LIST_REMOVE(dirrem, dm_next);
 		dirrem->dm_state |= COMPLETE;
 		dirrem->dm_dirinum = pagedep->pd_ino;
 		KASSERT(LIST_EMPTY(&dirrem->dm_jremrefhd),
 		    ("handle_written_filepage: Journal entries not written."));
 		add_to_worklist(&dirrem->dm_list, 0);
 	}
 	/*
 	 * Free any directory additions that have been committed.
 	 * If it is a newly allocated block, we have to wait until
 	 * the on-disk directory inode claims the new block.
 	 */
 	if ((pagedep->pd_state & NEWBLOCK) == 0)
 		while ((dap = LIST_FIRST(&pagedep->pd_pendinghd)) != NULL)
 			free_diradd(dap, NULL);
 	/*
 	 * Uncommitted directory entries must be restored.
 	 */
 	for (chgs = 0, i = 0; i < DAHASHSZ; i++) {
 		for (dap = LIST_FIRST(&pagedep->pd_diraddhd[i]); dap;
 		     dap = nextdap) {
 			nextdap = LIST_NEXT(dap, da_pdlist);
 			if (dap->da_state & ATTACHED)
 				panic("handle_written_filepage: attached");
 			ep = (struct direct *)
 			    ((char *)bp->b_data + dap->da_offset);
 			ep->d_ino = dap->da_newinum;
 			dap->da_state &= ~UNDONE;
 			dap->da_state |= ATTACHED;
 			chgs = 1;
 			/*
 			 * If the inode referenced by the directory has
 			 * been written out, then the dependency can be
 			 * moved to the pending list.
 			 */
 			if ((dap->da_state & ALLCOMPLETE) == ALLCOMPLETE) {
 				LIST_REMOVE(dap, da_pdlist);
 				LIST_INSERT_HEAD(&pagedep->pd_pendinghd, dap,
 				    da_pdlist);
 			}
 		}
 	}
 	/*
 	 * If there were any rollbacks in the directory, then it must be
 	 * marked dirty so that its will eventually get written back in
 	 * its correct form.
 	 */
 	if (chgs) {
 		if ((bp->b_flags & B_DELWRI) == 0)
 			stat_dir_entry++;
 		bdirty(bp);
 		return (1);
 	}
 	/*
 	 * If we are not waiting for a new directory block to be
 	 * claimed by its inode, then the pagedep will be freed.
 	 * Otherwise it will remain to track any new entries on
 	 * the page in case they are fsync'ed.
 	 */
 	free_pagedep(pagedep);
 	return (0);
 }
 
 /*
  * Writing back in-core inode structures.
  * 
  * The filesystem only accesses an inode's contents when it occupies an
  * "in-core" inode structure.  These "in-core" structures are separate from
  * the page frames used to cache inode blocks.  Only the latter are
  * transferred to/from the disk.  So, when the updated contents of the
  * "in-core" inode structure are copied to the corresponding in-memory inode
  * block, the dependencies are also transferred.  The following procedure is
  * called when copying a dirty "in-core" inode to a cached inode block.
  */
 
 /*
  * Called when an inode is loaded from disk. If the effective link count
  * differed from the actual link count when it was last flushed, then we
  * need to ensure that the correct effective link count is put back.
  */
 void 
 softdep_load_inodeblock(ip)
 	struct inode *ip;	/* the "in_core" copy of the inode */
 {
 	struct inodedep *inodedep;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_load_inodeblock called on non-softdep filesystem"));
 	/*
 	 * Check for alternate nlink count.
 	 */
 	ip->i_effnlink = ip->i_nlink;
 	ACQUIRE_LOCK(ip->i_ump);
 	if (inodedep_lookup(UFSTOVFS(ip->i_ump), ip->i_number, 0,
 	    &inodedep) == 0) {
 		FREE_LOCK(ip->i_ump);
 		return;
 	}
 	ip->i_effnlink -= inodedep->id_nlinkdelta;
 	FREE_LOCK(ip->i_ump);
 }
 
 /*
  * This routine is called just before the "in-core" inode
  * information is to be copied to the in-memory inode block.
  * Recall that an inode block contains several inodes. If
  * the force flag is set, then the dependencies will be
  * cleared so that the update can always be made. Note that
  * the buffer is locked when this routine is called, so we
  * will never be in the middle of writing the inode block 
  * to disk.
  */
 void 
 softdep_update_inodeblock(ip, bp, waitfor)
 	struct inode *ip;	/* the "in_core" copy of the inode */
 	struct buf *bp;		/* the buffer containing the inode block */
 	int waitfor;		/* nonzero => update must be allowed */
 {
 	struct inodedep *inodedep;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	struct worklist *wk;
 	struct mount *mp;
 	struct buf *ibp;
 	struct fs *fs;
 	int error;
 
 	ump = ip->i_ump;
 	mp = UFSTOVFS(ump);
 	KASSERT(MOUNTEDSOFTDEP(mp) != 0,
 	    ("softdep_update_inodeblock called on non-softdep filesystem"));
 	fs = ip->i_fs;
 	/*
 	 * Preserve the freelink that is on disk.  clear_unlinked_inodedep()
 	 * does not have access to the in-core ip so must write directly into
 	 * the inode block buffer when setting freelink.
 	 */
 	if (fs->fs_magic == FS_UFS1_MAGIC)
 		DIP_SET(ip, i_freelink, ((struct ufs1_dinode *)bp->b_data +
 		    ino_to_fsbo(fs, ip->i_number))->di_freelink);
 	else
 		DIP_SET(ip, i_freelink, ((struct ufs2_dinode *)bp->b_data +
 		    ino_to_fsbo(fs, ip->i_number))->di_freelink);
 	/*
 	 * If the effective link count is not equal to the actual link
 	 * count, then we must track the difference in an inodedep while
 	 * the inode is (potentially) tossed out of the cache. Otherwise,
 	 * if there is no existing inodedep, then there are no dependencies
 	 * to track.
 	 */
 	ACQUIRE_LOCK(ump);
 again:
 	if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) == 0) {
 		FREE_LOCK(ump);
 		if (ip->i_effnlink != ip->i_nlink)
 			panic("softdep_update_inodeblock: bad link count");
 		return;
 	}
 	if (inodedep->id_nlinkdelta != ip->i_nlink - ip->i_effnlink)
 		panic("softdep_update_inodeblock: bad delta");
 	/*
 	 * If we're flushing all dependencies we must also move any waiting
 	 * for journal writes onto the bufwait list prior to I/O.
 	 */
 	if (waitfor) {
 		TAILQ_FOREACH(inoref, &inodedep->id_inoreflst, if_deps) {
 			if ((inoref->if_state & (DEPCOMPLETE | GOINGAWAY))
 			    == DEPCOMPLETE) {
 				jwait(&inoref->if_list, MNT_WAIT);
 				goto again;
 			}
 		}
 	}
 	/*
 	 * Changes have been initiated. Anything depending on these
 	 * changes cannot occur until this inode has been written.
 	 */
 	inodedep->id_state &= ~COMPLETE;
 	if ((inodedep->id_state & ONWORKLIST) == 0)
 		WORKLIST_INSERT(&bp->b_dep, &inodedep->id_list);
 	/*
 	 * Any new dependencies associated with the incore inode must 
 	 * now be moved to the list associated with the buffer holding
 	 * the in-memory copy of the inode. Once merged process any
 	 * allocdirects that are completed by the merger.
 	 */
 	merge_inode_lists(&inodedep->id_newinoupdt, &inodedep->id_inoupdt);
 	if (!TAILQ_EMPTY(&inodedep->id_inoupdt))
 		handle_allocdirect_partdone(TAILQ_FIRST(&inodedep->id_inoupdt),
 		    NULL);
 	merge_inode_lists(&inodedep->id_newextupdt, &inodedep->id_extupdt);
 	if (!TAILQ_EMPTY(&inodedep->id_extupdt))
 		handle_allocdirect_partdone(TAILQ_FIRST(&inodedep->id_extupdt),
 		    NULL);
 	/*
 	 * Now that the inode has been pushed into the buffer, the
 	 * operations dependent on the inode being written to disk
 	 * can be moved to the id_bufwait so that they will be
 	 * processed when the buffer I/O completes.
 	 */
 	while ((wk = LIST_FIRST(&inodedep->id_inowait)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		WORKLIST_INSERT(&inodedep->id_bufwait, wk);
 	}
 	/*
 	 * Newly allocated inodes cannot be written until the bitmap
 	 * that allocates them have been written (indicated by
 	 * DEPCOMPLETE being set in id_state). If we are doing a
 	 * forced sync (e.g., an fsync on a file), we force the bitmap
 	 * to be written so that the update can be done.
 	 */
 	if (waitfor == 0) {
 		FREE_LOCK(ump);
 		return;
 	}
 retry:
 	if ((inodedep->id_state & (DEPCOMPLETE | GOINGAWAY)) != 0) {
 		FREE_LOCK(ump);
 		return;
 	}
 	ibp = inodedep->id_bmsafemap->sm_buf;
 	ibp = getdirtybuf(ibp, LOCK_PTR(ump), MNT_WAIT);
 	if (ibp == NULL) {
 		/*
 		 * If ibp came back as NULL, the dependency could have been
 		 * freed while we slept.  Look it up again, and check to see
 		 * that it has completed.
 		 */
 		if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) != 0)
 			goto retry;
 		FREE_LOCK(ump);
 		return;
 	}
 	FREE_LOCK(ump);
 	if ((error = bwrite(ibp)) != 0)
 		softdep_error("softdep_update_inodeblock: bwrite", error);
 }
 
 /*
  * Merge the a new inode dependency list (such as id_newinoupdt) into an
  * old inode dependency list (such as id_inoupdt). This routine must be
  * called with splbio interrupts blocked.
  */
 static void
 merge_inode_lists(newlisthead, oldlisthead)
 	struct allocdirectlst *newlisthead;
 	struct allocdirectlst *oldlisthead;
 {
 	struct allocdirect *listadp, *newadp;
 
 	newadp = TAILQ_FIRST(newlisthead);
 	for (listadp = TAILQ_FIRST(oldlisthead); listadp && newadp;) {
 		if (listadp->ad_offset < newadp->ad_offset) {
 			listadp = TAILQ_NEXT(listadp, ad_next);
 			continue;
 		}
 		TAILQ_REMOVE(newlisthead, newadp, ad_next);
 		TAILQ_INSERT_BEFORE(listadp, newadp, ad_next);
 		if (listadp->ad_offset == newadp->ad_offset) {
 			allocdirect_merge(oldlisthead, newadp,
 			    listadp);
 			listadp = newadp;
 		}
 		newadp = TAILQ_FIRST(newlisthead);
 	}
 	while ((newadp = TAILQ_FIRST(newlisthead)) != NULL) {
 		TAILQ_REMOVE(newlisthead, newadp, ad_next);
 		TAILQ_INSERT_TAIL(oldlisthead, newadp, ad_next);
 	}
 }
 
 /*
  * If we are doing an fsync, then we must ensure that any directory
  * entries for the inode have been written after the inode gets to disk.
  */
 int
 softdep_fsync(vp)
 	struct vnode *vp;	/* the "in_core" copy of the inode */
 {
 	struct inodedep *inodedep;
 	struct pagedep *pagedep;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	struct worklist *wk;
 	struct diradd *dap;
 	struct mount *mp;
 	struct vnode *pvp;
 	struct inode *ip;
 	struct buf *bp;
 	struct fs *fs;
 	struct thread *td = curthread;
 	int error, flushparent, pagedep_new_block;
 	ino_t parentino;
 	ufs_lbn_t lbn;
 
 	ip = VTOI(vp);
 	fs = ip->i_fs;
 	ump = ip->i_ump;
 	mp = vp->v_mount;
 	if (MOUNTEDSOFTDEP(mp) == 0)
 		return (0);
 	ACQUIRE_LOCK(ump);
 restart:
 	if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) == 0) {
 		FREE_LOCK(ump);
 		return (0);
 	}
 	TAILQ_FOREACH(inoref, &inodedep->id_inoreflst, if_deps) {
 		if ((inoref->if_state & (DEPCOMPLETE | GOINGAWAY))
 		    == DEPCOMPLETE) {
 			jwait(&inoref->if_list, MNT_WAIT);
 			goto restart;
 		}
 	}
 	if (!LIST_EMPTY(&inodedep->id_inowait) ||
 	    !TAILQ_EMPTY(&inodedep->id_extupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_newextupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_inoupdt) ||
 	    !TAILQ_EMPTY(&inodedep->id_newinoupdt))
 		panic("softdep_fsync: pending ops %p", inodedep);
 	for (error = 0, flushparent = 0; ; ) {
 		if ((wk = LIST_FIRST(&inodedep->id_pendinghd)) == NULL)
 			break;
 		if (wk->wk_type != D_DIRADD)
 			panic("softdep_fsync: Unexpected type %s",
 			    TYPENAME(wk->wk_type));
 		dap = WK_DIRADD(wk);
 		/*
 		 * Flush our parent if this directory entry has a MKDIR_PARENT
 		 * dependency or is contained in a newly allocated block.
 		 */
 		if (dap->da_state & DIRCHG)
 			pagedep = dap->da_previous->dm_pagedep;
 		else
 			pagedep = dap->da_pagedep;
 		parentino = pagedep->pd_ino;
 		lbn = pagedep->pd_lbn;
 		if ((dap->da_state & (MKDIR_BODY | COMPLETE)) != COMPLETE)
 			panic("softdep_fsync: dirty");
 		if ((dap->da_state & MKDIR_PARENT) ||
 		    (pagedep->pd_state & NEWBLOCK))
 			flushparent = 1;
 		else
 			flushparent = 0;
 		/*
 		 * If we are being fsync'ed as part of vgone'ing this vnode,
 		 * then we will not be able to release and recover the
 		 * vnode below, so we just have to give up on writing its
 		 * directory entry out. It will eventually be written, just
 		 * not now, but then the user was not asking to have it
 		 * written, so we are not breaking any promises.
 		 */
 		if (vp->v_iflag & VI_DOOMED)
 			break;
 		/*
 		 * We prevent deadlock by always fetching inodes from the
 		 * root, moving down the directory tree. Thus, when fetching
 		 * our parent directory, we first try to get the lock. If
 		 * that fails, we must unlock ourselves before requesting
 		 * the lock on our parent. See the comment in ufs_lookup
 		 * for details on possible races.
 		 */
 		FREE_LOCK(ump);
 		if (ffs_vgetf(mp, parentino, LK_NOWAIT | LK_EXCLUSIVE, &pvp,
 		    FFSV_FORCEINSMQ)) {
 			error = vfs_busy(mp, MBF_NOWAIT);
 			if (error != 0) {
 				vfs_ref(mp);
 				VOP_UNLOCK(vp, 0);
 				error = vfs_busy(mp, 0);
 				vn_lock(vp, LK_EXCLUSIVE | LK_RETRY);
 				vfs_rel(mp);
 				if (error != 0)
 					return (ENOENT);
 				if (vp->v_iflag & VI_DOOMED) {
 					vfs_unbusy(mp);
 					return (ENOENT);
 				}
 			}
 			VOP_UNLOCK(vp, 0);
 			error = ffs_vgetf(mp, parentino, LK_EXCLUSIVE,
 			    &pvp, FFSV_FORCEINSMQ);
 			vfs_unbusy(mp);
 			vn_lock(vp, LK_EXCLUSIVE | LK_RETRY);
 			if (vp->v_iflag & VI_DOOMED) {
 				if (error == 0)
 					vput(pvp);
 				error = ENOENT;
 			}
 			if (error != 0)
 				return (error);
 		}
 		/*
 		 * All MKDIR_PARENT dependencies and all the NEWBLOCK pagedeps
 		 * that are contained in direct blocks will be resolved by 
 		 * doing a ffs_update. Pagedeps contained in indirect blocks
 		 * may require a complete sync'ing of the directory. So, we
 		 * try the cheap and fast ffs_update first, and if that fails,
 		 * then we do the slower ffs_syncvnode of the directory.
 		 */
 		if (flushparent) {
 			int locked;
 
 			if ((error = ffs_update(pvp, 1)) != 0) {
 				vput(pvp);
 				return (error);
 			}
 			ACQUIRE_LOCK(ump);
 			locked = 1;
 			if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) != 0) {
 				if ((wk = LIST_FIRST(&inodedep->id_pendinghd)) != NULL) {
 					if (wk->wk_type != D_DIRADD)
 						panic("softdep_fsync: Unexpected type %s",
 						      TYPENAME(wk->wk_type));
 					dap = WK_DIRADD(wk);
 					if (dap->da_state & DIRCHG)
 						pagedep = dap->da_previous->dm_pagedep;
 					else
 						pagedep = dap->da_pagedep;
 					pagedep_new_block = pagedep->pd_state & NEWBLOCK;
 					FREE_LOCK(ump);
 					locked = 0;
 					if (pagedep_new_block && (error =
 					    ffs_syncvnode(pvp, MNT_WAIT, 0))) {
 						vput(pvp);
 						return (error);
 					}
 				}
 			}
 			if (locked)
 				FREE_LOCK(ump);
 		}
 		/*
 		 * Flush directory page containing the inode's name.
 		 */
 		error = bread(pvp, lbn, blksize(fs, VTOI(pvp), lbn), td->td_ucred,
 		    &bp);
 		if (error == 0)
 			error = bwrite(bp);
 		else
 			brelse(bp);
 		vput(pvp);
 		if (error != 0)
 			return (error);
 		ACQUIRE_LOCK(ump);
 		if (inodedep_lookup(mp, ip->i_number, 0, &inodedep) == 0)
 			break;
 	}
 	FREE_LOCK(ump);
 	return (0);
 }
 
 /*
  * Flush all the dirty bitmaps associated with the block device
  * before flushing the rest of the dirty blocks so as to reduce
  * the number of dependencies that will have to be rolled back.
  *
  * XXX Unused?
  */
 void
 softdep_fsync_mountdev(vp)
 	struct vnode *vp;
 {
 	struct buf *bp, *nbp;
 	struct worklist *wk;
 	struct bufobj *bo;
 
 	if (!vn_isdisk(vp, NULL))
 		panic("softdep_fsync_mountdev: vnode not a disk");
 	bo = &vp->v_bufobj;
 restart:
 	BO_LOCK(bo);
 	TAILQ_FOREACH_SAFE(bp, &bo->bo_dirty.bv_hd, b_bobufs, nbp) {
 		/* 
 		 * If it is already scheduled, skip to the next buffer.
 		 */
 		if (BUF_LOCK(bp, LK_EXCLUSIVE | LK_NOWAIT, NULL))
 			continue;
 
 		if ((bp->b_flags & B_DELWRI) == 0)
 			panic("softdep_fsync_mountdev: not dirty");
 		/*
 		 * We are only interested in bitmaps with outstanding
 		 * dependencies.
 		 */
 		if ((wk = LIST_FIRST(&bp->b_dep)) == NULL ||
 		    wk->wk_type != D_BMSAFEMAP ||
 		    (bp->b_vflags & BV_BKGRDINPROG)) {
 			BUF_UNLOCK(bp);
 			continue;
 		}
 		BO_UNLOCK(bo);
 		bremfree(bp);
 		(void) bawrite(bp);
 		goto restart;
 	}
 	drain_output(vp);
 	BO_UNLOCK(bo);
 }
 
 /*
  * Sync all cylinder groups that were dirty at the time this function is
  * called.  Newly dirtied cgs will be inserted before the sentinel.  This
  * is used to flush freedep activity that may be holding up writes to a
  * indirect block.
  */
 static int
 sync_cgs(mp, waitfor)
 	struct mount *mp;
 	int waitfor;
 {
 	struct bmsafemap *bmsafemap;
 	struct bmsafemap *sentinel;
 	struct ufsmount *ump;
 	struct buf *bp;
 	int error;
 
 	sentinel = malloc(sizeof(*sentinel), M_BMSAFEMAP, M_ZERO | M_WAITOK);
 	sentinel->sm_cg = -1;
 	ump = VFSTOUFS(mp);
 	error = 0;
 	ACQUIRE_LOCK(ump);
 	LIST_INSERT_HEAD(&ump->softdep_dirtycg, sentinel, sm_next);
 	for (bmsafemap = LIST_NEXT(sentinel, sm_next); bmsafemap != NULL;
 	    bmsafemap = LIST_NEXT(sentinel, sm_next)) {
 		/* Skip sentinels and cgs with no work to release. */
 		if (bmsafemap->sm_cg == -1 ||
 		    (LIST_EMPTY(&bmsafemap->sm_freehd) &&
 		    LIST_EMPTY(&bmsafemap->sm_freewr))) {
 			LIST_REMOVE(sentinel, sm_next);
 			LIST_INSERT_AFTER(bmsafemap, sentinel, sm_next);
 			continue;
 		}
 		/*
 		 * If we don't get the lock and we're waiting try again, if
 		 * not move on to the next buf and try to sync it.
 		 */
 		bp = getdirtybuf(bmsafemap->sm_buf, LOCK_PTR(ump), waitfor);
 		if (bp == NULL && waitfor == MNT_WAIT)
 			continue;
 		LIST_REMOVE(sentinel, sm_next);
 		LIST_INSERT_AFTER(bmsafemap, sentinel, sm_next);
 		if (bp == NULL)
 			continue;
 		FREE_LOCK(ump);
 		if (waitfor == MNT_NOWAIT)
 			bawrite(bp);
 		else
 			error = bwrite(bp);
 		ACQUIRE_LOCK(ump);
 		if (error)
 			break;
 	}
 	LIST_REMOVE(sentinel, sm_next);
 	FREE_LOCK(ump);
 	free(sentinel, M_BMSAFEMAP);
 	return (error);
 }
 
 /*
  * This routine is called when we are trying to synchronously flush a
  * file. This routine must eliminate any filesystem metadata dependencies
  * so that the syncing routine can succeed.
  */
 int
 softdep_sync_metadata(struct vnode *vp)
 {
 	struct inode *ip;
 	int error;
 
 	ip = VTOI(vp);
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_sync_metadata called on non-softdep filesystem"));
 	/*
 	 * Ensure that any direct block dependencies have been cleared,
 	 * truncations are started, and inode references are journaled.
 	 */
 	ACQUIRE_LOCK(ip->i_ump);
 	/*
 	 * Write all journal records to prevent rollbacks on devvp.
 	 */
 	if (vp->v_type == VCHR)
 		softdep_flushjournal(vp->v_mount);
 	error = flush_inodedep_deps(vp, vp->v_mount, ip->i_number);
 	/*
 	 * Ensure that all truncates are written so we won't find deps on
 	 * indirect blocks.
 	 */
 	process_truncates(vp);
 	FREE_LOCK(ip->i_ump);
 
 	return (error);
 }
 
 /*
  * This routine is called when we are attempting to sync a buf with
  * dependencies.  If waitfor is MNT_NOWAIT it attempts to schedule any
  * other IO it can but returns EBUSY if the buffer is not yet able to
  * be written.  Dependencies which will not cause rollbacks will always
  * return 0.
  */
 int
 softdep_sync_buf(struct vnode *vp, struct buf *bp, int waitfor)
 {
 	struct indirdep *indirdep;
 	struct pagedep *pagedep;
 	struct allocindir *aip;
 	struct newblk *newblk;
 	struct ufsmount *ump;
 	struct buf *nbp;
 	struct worklist *wk;
 	int i, error;
 
 	KASSERT(MOUNTEDSOFTDEP(vp->v_mount) != 0,
 	    ("softdep_sync_buf called on non-softdep filesystem"));
 	/*
 	 * For VCHR we just don't want to force flush any dependencies that
 	 * will cause rollbacks.
 	 */
 	if (vp->v_type == VCHR) {
 		if (waitfor == MNT_NOWAIT && softdep_count_dependencies(bp, 0))
 			return (EBUSY);
 		return (0);
 	}
 	ump = VTOI(vp)->i_ump;
 	ACQUIRE_LOCK(ump);
 	/*
 	 * As we hold the buffer locked, none of its dependencies
 	 * will disappear.
 	 */
 	error = 0;
 top:
 	LIST_FOREACH(wk, &bp->b_dep, wk_list) {
 		switch (wk->wk_type) {
 
 		case D_ALLOCDIRECT:
 		case D_ALLOCINDIR:
 			newblk = WK_NEWBLK(wk);
 			if (newblk->nb_jnewblk != NULL) {
 				if (waitfor == MNT_NOWAIT) {
 					error = EBUSY;
 					goto out_unlock;
 				}
 				jwait(&newblk->nb_jnewblk->jn_list, waitfor);
 				goto top;
 			}
 			if (newblk->nb_state & DEPCOMPLETE ||
 			    waitfor == MNT_NOWAIT)
 				continue;
 			nbp = newblk->nb_bmsafemap->sm_buf;
 			nbp = getdirtybuf(nbp, LOCK_PTR(ump), waitfor);
 			if (nbp == NULL)
 				goto top;
 			FREE_LOCK(ump);
 			if ((error = bwrite(nbp)) != 0)
 				goto out;
 			ACQUIRE_LOCK(ump);
 			continue;
 
 		case D_INDIRDEP:
 			indirdep = WK_INDIRDEP(wk);
 			if (waitfor == MNT_NOWAIT) {
 				if (!TAILQ_EMPTY(&indirdep->ir_trunc) ||
 				    !LIST_EMPTY(&indirdep->ir_deplisthd)) {
 					error = EBUSY;
 					goto out_unlock;
 				}
 			}
 			if (!TAILQ_EMPTY(&indirdep->ir_trunc))
 				panic("softdep_sync_buf: truncation pending.");
 		restart:
 			LIST_FOREACH(aip, &indirdep->ir_deplisthd, ai_next) {
 				newblk = (struct newblk *)aip;
 				if (newblk->nb_jnewblk != NULL) {
 					jwait(&newblk->nb_jnewblk->jn_list,
 					    waitfor);
 					goto restart;
 				}
 				if (newblk->nb_state & DEPCOMPLETE)
 					continue;
 				nbp = newblk->nb_bmsafemap->sm_buf;
 				nbp = getdirtybuf(nbp, LOCK_PTR(ump), waitfor);
 				if (nbp == NULL)
 					goto restart;
 				FREE_LOCK(ump);
 				if ((error = bwrite(nbp)) != 0)
 					goto out;
 				ACQUIRE_LOCK(ump);
 				goto restart;
 			}
 			continue;
 
 		case D_PAGEDEP:
 			/*
 			 * Only flush directory entries in synchronous passes.
 			 */
 			if (waitfor != MNT_WAIT) {
 				error = EBUSY;
 				goto out_unlock;
 			}
 			/*
 			 * While syncing snapshots, we must allow recursive
 			 * lookups.
 			 */
 			BUF_AREC(bp);
 			/*
 			 * We are trying to sync a directory that may
 			 * have dependencies on both its own metadata
 			 * and/or dependencies on the inodes of any
 			 * recently allocated files. We walk its diradd
 			 * lists pushing out the associated inode.
 			 */
 			pagedep = WK_PAGEDEP(wk);
 			for (i = 0; i < DAHASHSZ; i++) {
 				if (LIST_FIRST(&pagedep->pd_diraddhd[i]) == 0)
 					continue;
 				if ((error = flush_pagedep_deps(vp, wk->wk_mp,
 				    &pagedep->pd_diraddhd[i]))) {
 					BUF_NOREC(bp);
 					goto out_unlock;
 				}
 			}
 			BUF_NOREC(bp);
 			continue;
 
 		case D_FREEWORK:
 		case D_FREEDEP:
 		case D_JSEGDEP:
 		case D_JNEWBLK:
 			continue;
 
 		default:
 			panic("softdep_sync_buf: Unknown type %s",
 			    TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 	}
 out_unlock:
 	FREE_LOCK(ump);
 out:
 	return (error);
 }
 
 /*
  * Flush the dependencies associated with an inodedep.
  * Called with splbio blocked.
  */
 static int
 flush_inodedep_deps(vp, mp, ino)
 	struct vnode *vp;
 	struct mount *mp;
 	ino_t ino;
 {
 	struct inodedep *inodedep;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	int error, waitfor;
 
 	/*
 	 * This work is done in two passes. The first pass grabs most
 	 * of the buffers and begins asynchronously writing them. The
 	 * only way to wait for these asynchronous writes is to sleep
 	 * on the filesystem vnode which may stay busy for a long time
 	 * if the filesystem is active. So, instead, we make a second
 	 * pass over the dependencies blocking on each write. In the
 	 * usual case we will be blocking against a write that we
 	 * initiated, so when it is done the dependency will have been
 	 * resolved. Thus the second pass is expected to end quickly.
 	 * We give a brief window at the top of the loop to allow
 	 * any pending I/O to complete.
 	 */
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	for (error = 0, waitfor = MNT_NOWAIT; ; ) {
 		if (error)
 			return (error);
 		FREE_LOCK(ump);
 		ACQUIRE_LOCK(ump);
 restart:
 		if (inodedep_lookup(mp, ino, 0, &inodedep) == 0)
 			return (0);
 		TAILQ_FOREACH(inoref, &inodedep->id_inoreflst, if_deps) {
 			if ((inoref->if_state & (DEPCOMPLETE | GOINGAWAY))
 			    == DEPCOMPLETE) {
 				jwait(&inoref->if_list, MNT_WAIT);
 				goto restart;
 			}
 		}
 		if (flush_deplist(&inodedep->id_inoupdt, waitfor, &error) ||
 		    flush_deplist(&inodedep->id_newinoupdt, waitfor, &error) ||
 		    flush_deplist(&inodedep->id_extupdt, waitfor, &error) ||
 		    flush_deplist(&inodedep->id_newextupdt, waitfor, &error))
 			continue;
 		/*
 		 * If pass2, we are done, otherwise do pass 2.
 		 */
 		if (waitfor == MNT_WAIT)
 			break;
 		waitfor = MNT_WAIT;
 	}
 	/*
 	 * Try freeing inodedep in case all dependencies have been removed.
 	 */
 	if (inodedep_lookup(mp, ino, 0, &inodedep) != 0)
 		(void) free_inodedep(inodedep);
 	return (0);
 }
 
 /*
  * Flush an inode dependency list.
  * Called with splbio blocked.
  */
 static int
 flush_deplist(listhead, waitfor, errorp)
 	struct allocdirectlst *listhead;
 	int waitfor;
 	int *errorp;
 {
 	struct allocdirect *adp;
 	struct newblk *newblk;
 	struct ufsmount *ump;
 	struct buf *bp;
 
 	if ((adp = TAILQ_FIRST(listhead)) == NULL)
 		return (0);
 	ump = VFSTOUFS(adp->ad_list.wk_mp);
 	LOCK_OWNED(ump);
 	TAILQ_FOREACH(adp, listhead, ad_next) {
 		newblk = (struct newblk *)adp;
 		if (newblk->nb_jnewblk != NULL) {
 			jwait(&newblk->nb_jnewblk->jn_list, MNT_WAIT);
 			return (1);
 		}
 		if (newblk->nb_state & DEPCOMPLETE)
 			continue;
 		bp = newblk->nb_bmsafemap->sm_buf;
 		bp = getdirtybuf(bp, LOCK_PTR(ump), waitfor);
 		if (bp == NULL) {
 			if (waitfor == MNT_NOWAIT)
 				continue;
 			return (1);
 		}
 		FREE_LOCK(ump);
 		if (waitfor == MNT_NOWAIT)
 			bawrite(bp);
 		else 
 			*errorp = bwrite(bp);
 		ACQUIRE_LOCK(ump);
 		return (1);
 	}
 	return (0);
 }
 
 /*
  * Flush dependencies associated with an allocdirect block.
  */
 static int
 flush_newblk_dep(vp, mp, lbn)
 	struct vnode *vp;
 	struct mount *mp;
 	ufs_lbn_t lbn;
 {
 	struct newblk *newblk;
 	struct ufsmount *ump;
 	struct bufobj *bo;
 	struct inode *ip;
 	struct buf *bp;
 	ufs2_daddr_t blkno;
 	int error;
 
 	error = 0;
 	bo = &vp->v_bufobj;
 	ip = VTOI(vp);
 	blkno = DIP(ip, i_db[lbn]);
 	if (blkno == 0)
 		panic("flush_newblk_dep: Missing block");
 	ump = VFSTOUFS(mp);
 	ACQUIRE_LOCK(ump);
 	/*
 	 * Loop until all dependencies related to this block are satisfied.
 	 * We must be careful to restart after each sleep in case a write
 	 * completes some part of this process for us.
 	 */
 	for (;;) {
 		if (newblk_lookup(mp, blkno, 0, &newblk) == 0) {
 			FREE_LOCK(ump);
 			break;
 		}
 		if (newblk->nb_list.wk_type != D_ALLOCDIRECT)
 			panic("flush_newblk_deps: Bad newblk %p", newblk);
 		/*
 		 * Flush the journal.
 		 */
 		if (newblk->nb_jnewblk != NULL) {
 			jwait(&newblk->nb_jnewblk->jn_list, MNT_WAIT);
 			continue;
 		}
 		/*
 		 * Write the bitmap dependency.
 		 */
 		if ((newblk->nb_state & DEPCOMPLETE) == 0) {
 			bp = newblk->nb_bmsafemap->sm_buf;
 			bp = getdirtybuf(bp, LOCK_PTR(ump), MNT_WAIT);
 			if (bp == NULL)
 				continue;
 			FREE_LOCK(ump);
 			error = bwrite(bp);
 			if (error)
 				break;
 			ACQUIRE_LOCK(ump);
 			continue;
 		}
 		/*
 		 * Write the buffer.
 		 */
 		FREE_LOCK(ump);
 		BO_LOCK(bo);
 		bp = gbincore(bo, lbn);
 		if (bp != NULL) {
 			error = BUF_LOCK(bp, LK_EXCLUSIVE | LK_SLEEPFAIL |
 			    LK_INTERLOCK, BO_LOCKPTR(bo));
 			if (error == ENOLCK) {
 				ACQUIRE_LOCK(ump);
 				continue; /* Slept, retry */
 			}
 			if (error != 0)
 				break;	/* Failed */
 			if (bp->b_flags & B_DELWRI) {
 				bremfree(bp);
 				error = bwrite(bp);
 				if (error)
 					break;
 			} else
 				BUF_UNLOCK(bp);
 		} else
 			BO_UNLOCK(bo);
 		/*
 		 * We have to wait for the direct pointers to
 		 * point at the newdirblk before the dependency
 		 * will go away.
 		 */
 		error = ffs_update(vp, 1);
 		if (error)
 			break;
 		ACQUIRE_LOCK(ump);
 	}
 	return (error);
 }
 
 /*
  * Eliminate a pagedep dependency by flushing out all its diradd dependencies.
  * Called with splbio blocked.
  */
 static int
 flush_pagedep_deps(pvp, mp, diraddhdp)
 	struct vnode *pvp;
 	struct mount *mp;
 	struct diraddhd *diraddhdp;
 {
 	struct inodedep *inodedep;
 	struct inoref *inoref;
 	struct ufsmount *ump;
 	struct diradd *dap;
 	struct vnode *vp;
 	int error = 0;
 	struct buf *bp;
 	ino_t inum;
 	struct diraddhd unfinished;
 
 	LIST_INIT(&unfinished);
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 restart:
 	while ((dap = LIST_FIRST(diraddhdp)) != NULL) {
 		/*
 		 * Flush ourselves if this directory entry
 		 * has a MKDIR_PARENT dependency.
 		 */
 		if (dap->da_state & MKDIR_PARENT) {
 			FREE_LOCK(ump);
 			if ((error = ffs_update(pvp, 1)) != 0)
 				break;
 			ACQUIRE_LOCK(ump);
 			/*
 			 * If that cleared dependencies, go on to next.
 			 */
 			if (dap != LIST_FIRST(diraddhdp))
 				continue;
 			/*
 			 * All MKDIR_PARENT dependencies and all the
 			 * NEWBLOCK pagedeps that are contained in direct
 			 * blocks were resolved by doing above ffs_update.
 			 * Pagedeps contained in indirect blocks may
 			 * require a complete sync'ing of the directory.
 			 * We are in the midst of doing a complete sync,
 			 * so if they are not resolved in this pass we
 			 * defer them for now as they will be sync'ed by
 			 * our caller shortly.
 			 */
 			LIST_REMOVE(dap, da_pdlist);
 			LIST_INSERT_HEAD(&unfinished, dap, da_pdlist);
 			continue;
 		}
 		/*
 		 * A newly allocated directory must have its "." and
 		 * ".." entries written out before its name can be
 		 * committed in its parent. 
 		 */
 		inum = dap->da_newinum;
 		if (inodedep_lookup(UFSTOVFS(ump), inum, 0, &inodedep) == 0)
 			panic("flush_pagedep_deps: lost inode1");
 		/*
 		 * Wait for any pending journal adds to complete so we don't
 		 * cause rollbacks while syncing.
 		 */
 		TAILQ_FOREACH(inoref, &inodedep->id_inoreflst, if_deps) {
 			if ((inoref->if_state & (DEPCOMPLETE | GOINGAWAY))
 			    == DEPCOMPLETE) {
 				jwait(&inoref->if_list, MNT_WAIT);
 				goto restart;
 			}
 		}
 		if (dap->da_state & MKDIR_BODY) {
 			FREE_LOCK(ump);
 			if ((error = ffs_vgetf(mp, inum, LK_EXCLUSIVE, &vp,
 			    FFSV_FORCEINSMQ)))
 				break;
 			error = flush_newblk_dep(vp, mp, 0);
 			/*
 			 * If we still have the dependency we might need to
 			 * update the vnode to sync the new link count to
 			 * disk.
 			 */
 			if (error == 0 && dap == LIST_FIRST(diraddhdp))
 				error = ffs_update(vp, 1);
 			vput(vp);
 			if (error != 0)
 				break;
 			ACQUIRE_LOCK(ump);
 			/*
 			 * If that cleared dependencies, go on to next.
 			 */
 			if (dap != LIST_FIRST(diraddhdp))
 				continue;
 			if (dap->da_state & MKDIR_BODY) {
 				inodedep_lookup(UFSTOVFS(ump), inum, 0,
 				    &inodedep);
 				panic("flush_pagedep_deps: MKDIR_BODY "
 				    "inodedep %p dap %p vp %p",
 				    inodedep, dap, vp);
 			}
 		}
 		/*
 		 * Flush the inode on which the directory entry depends.
 		 * Having accounted for MKDIR_PARENT and MKDIR_BODY above,
 		 * the only remaining dependency is that the updated inode
 		 * count must get pushed to disk. The inode has already
 		 * been pushed into its inode buffer (via VOP_UPDATE) at
 		 * the time of the reference count change. So we need only
 		 * locate that buffer, ensure that there will be no rollback
 		 * caused by a bitmap dependency, then write the inode buffer.
 		 */
 retry:
 		if (inodedep_lookup(UFSTOVFS(ump), inum, 0, &inodedep) == 0)
 			panic("flush_pagedep_deps: lost inode");
 		/*
 		 * If the inode still has bitmap dependencies,
 		 * push them to disk.
 		 */
 		if ((inodedep->id_state & (DEPCOMPLETE | GOINGAWAY)) == 0) {
 			bp = inodedep->id_bmsafemap->sm_buf;
 			bp = getdirtybuf(bp, LOCK_PTR(ump), MNT_WAIT);
 			if (bp == NULL)
 				goto retry;
 			FREE_LOCK(ump);
 			if ((error = bwrite(bp)) != 0)
 				break;
 			ACQUIRE_LOCK(ump);
 			if (dap != LIST_FIRST(diraddhdp))
 				continue;
 		}
 		/*
 		 * If the inode is still sitting in a buffer waiting
 		 * to be written or waiting for the link count to be
 		 * adjusted update it here to flush it to disk.
 		 */
 		if (dap == LIST_FIRST(diraddhdp)) {
 			FREE_LOCK(ump);
 			if ((error = ffs_vgetf(mp, inum, LK_EXCLUSIVE, &vp,
 			    FFSV_FORCEINSMQ)))
 				break;
 			error = ffs_update(vp, 1);
 			vput(vp);
 			if (error)
 				break;
 			ACQUIRE_LOCK(ump);
 		}
 		/*
 		 * If we have failed to get rid of all the dependencies
 		 * then something is seriously wrong.
 		 */
 		if (dap == LIST_FIRST(diraddhdp)) {
 			inodedep_lookup(UFSTOVFS(ump), inum, 0, &inodedep);
 			panic("flush_pagedep_deps: failed to flush " 
 			    "inodedep %p ino %ju dap %p",
 			    inodedep, (uintmax_t)inum, dap);
 		}
 	}
 	if (error)
 		ACQUIRE_LOCK(ump);
 	while ((dap = LIST_FIRST(&unfinished)) != NULL) {
 		LIST_REMOVE(dap, da_pdlist);
 		LIST_INSERT_HEAD(diraddhdp, dap, da_pdlist);
 	}
 	return (error);
 }
 
 /*
  * A large burst of file addition or deletion activity can drive the
  * memory load excessively high. First attempt to slow things down
  * using the techniques below. If that fails, this routine requests
  * the offending operations to fall back to running synchronously
  * until the memory load returns to a reasonable level.
  */
 int
 softdep_slowdown(vp)
 	struct vnode *vp;
 {
 	struct ufsmount *ump;
 	int jlow;
 	int max_softdeps_hard;
 
 	KASSERT(MOUNTEDSOFTDEP(vp->v_mount) != 0,
 	    ("softdep_slowdown called on non-softdep filesystem"));
 	ump = VFSTOUFS(vp->v_mount);
 	ACQUIRE_LOCK(ump);
 	jlow = 0;
 	/*
 	 * Check for journal space if needed.
 	 */
 	if (DOINGSUJ(vp)) {
 		if (journal_space(ump, 0) == 0)
 			jlow = 1;
 	}
 	/*
 	 * If the system is under its limits and our filesystem is
 	 * not responsible for more than our share of the usage and
 	 * we are not low on journal space, then no need to slow down.
 	 */
 	max_softdeps_hard = max_softdeps * 11 / 10;
 	if (dep_current[D_DIRREM] < max_softdeps_hard / 2 &&
 	    dep_current[D_INODEDEP] < max_softdeps_hard &&
 	    dep_current[D_INDIRDEP] < max_softdeps_hard / 1000 &&
 	    dep_current[D_FREEBLKS] < max_softdeps_hard && jlow == 0 &&
 	    ump->softdep_curdeps[D_DIRREM] <
 	    (max_softdeps_hard / 2) / stat_flush_threads &&
 	    ump->softdep_curdeps[D_INODEDEP] <
 	    max_softdeps_hard / stat_flush_threads &&
 	    ump->softdep_curdeps[D_INDIRDEP] <
 	    (max_softdeps_hard / 1000) / stat_flush_threads &&
 	    ump->softdep_curdeps[D_FREEBLKS] <
 	    max_softdeps_hard / stat_flush_threads) {
 		FREE_LOCK(ump);
   		return (0);
 	}
 	/*
 	 * If the journal is low or our filesystem is over its limit
 	 * then speedup the cleanup.
 	 */
 	if (ump->softdep_curdeps[D_INDIRDEP] <
 	    (max_softdeps_hard / 1000) / stat_flush_threads || jlow)
 		softdep_speedup(ump);
 	stat_sync_limit_hit += 1;
 	FREE_LOCK(ump);
 	/*
 	 * We only slow down the rate at which new dependencies are
 	 * generated if we are not using journaling. With journaling,
 	 * the cleanup should always be sufficient to keep things
 	 * under control.
 	 */
 	if (DOINGSUJ(vp))
 		return (0);
 	return (1);
 }
 
 /*
  * Called by the allocation routines when they are about to fail
  * in the hope that we can free up the requested resource (inodes
  * or disk space).
  * 
  * First check to see if the work list has anything on it. If it has,
  * clean up entries until we successfully free the requested resource.
  * Because this process holds inodes locked, we cannot handle any remove
  * requests that might block on a locked inode as that could lead to
  * deadlock. If the worklist yields none of the requested resource,
  * start syncing out vnodes to free up the needed space.
  */
 int
 softdep_request_cleanup(fs, vp, cred, resource)
 	struct fs *fs;
 	struct vnode *vp;
 	struct ucred *cred;
 	int resource;
 {
 	struct ufsmount *ump;
 	struct mount *mp;
 	struct vnode *lvp, *mvp;
 	long starttime;
 	ufs2_daddr_t needed;
 	int error;
 
 	/*
 	 * If we are being called because of a process doing a
 	 * copy-on-write, then it is not safe to process any
 	 * worklist items as we will recurse into the copyonwrite
 	 * routine.  This will result in an incoherent snapshot.
 	 * If the vnode that we hold is a snapshot, we must avoid
 	 * handling other resources that could cause deadlock.
 	 */
 	if ((curthread->td_pflags & TDP_COWINPROGRESS) || IS_SNAPSHOT(VTOI(vp)))
 		return (0);
 
 	if (resource == FLUSH_BLOCKS_WAIT)
 		stat_cleanup_blkrequests += 1;
 	else
 		stat_cleanup_inorequests += 1;
 
 	mp = vp->v_mount;
 	ump = VFSTOUFS(mp);
 	mtx_assert(UFS_MTX(ump), MA_OWNED);
 	UFS_UNLOCK(ump);
 	error = ffs_update(vp, 1);
 	if (error != 0 || MOUNTEDSOFTDEP(mp) == 0) {
 		UFS_LOCK(ump);
 		return (0);
 	}
 	/*
 	 * If we are in need of resources, start by cleaning up
 	 * any block removals associated with our inode.
 	 */
 	ACQUIRE_LOCK(ump);
 	process_removes(vp);
 	process_truncates(vp);
 	FREE_LOCK(ump);
 	/*
 	 * Now clean up at least as many resources as we will need.
 	 *
 	 * When requested to clean up inodes, the number that are needed
 	 * is set by the number of simultaneous writers (mnt_writeopcount)
 	 * plus a bit of slop (2) in case some more writers show up while
 	 * we are cleaning.
 	 *
 	 * When requested to free up space, the amount of space that
 	 * we need is enough blocks to allocate a full-sized segment
 	 * (fs_contigsumsize). The number of such segments that will
 	 * be needed is set by the number of simultaneous writers
 	 * (mnt_writeopcount) plus a bit of slop (2) in case some more
 	 * writers show up while we are cleaning.
 	 *
 	 * Additionally, if we are unpriviledged and allocating space,
 	 * we need to ensure that we clean up enough blocks to get the
 	 * needed number of blocks over the threshhold of the minimum
 	 * number of blocks required to be kept free by the filesystem
 	 * (fs_minfree).
 	 */
 	if (resource == FLUSH_INODES_WAIT) {
 		needed = vp->v_mount->mnt_writeopcount + 2;
 	} else if (resource == FLUSH_BLOCKS_WAIT) {
 		needed = (vp->v_mount->mnt_writeopcount + 2) *
 		    fs->fs_contigsumsize;
 		if (priv_check_cred(cred, PRIV_VFS_BLOCKRESERVE, 0))
 			needed += fragstoblks(fs,
 			    roundup((fs->fs_dsize * fs->fs_minfree / 100) -
 			    fs->fs_cstotal.cs_nffree, fs->fs_frag));
 	} else {
 		UFS_LOCK(ump);
 		printf("softdep_request_cleanup: Unknown resource type %d\n",
 		    resource);
 		return (0);
 	}
 	starttime = time_second;
 retry:
 	if ((resource == FLUSH_BLOCKS_WAIT && ump->softdep_on_worklist > 0 &&
 	    fs->fs_cstotal.cs_nbfree <= needed) ||
 	    (resource == FLUSH_INODES_WAIT && fs->fs_pendinginodes > 0 &&
 	    fs->fs_cstotal.cs_nifree <= needed)) {
 		ACQUIRE_LOCK(ump);
 		if (ump->softdep_on_worklist > 0 &&
 		    process_worklist_item(UFSTOVFS(ump),
 		    ump->softdep_on_worklist, LK_NOWAIT) != 0)
 			stat_worklist_push += 1;
 		FREE_LOCK(ump);
 	}
 	/*
 	 * If we still need resources and there are no more worklist
 	 * entries to process to obtain them, we have to start flushing
 	 * the dirty vnodes to force the release of additional requests
 	 * to the worklist that we can then process to reap addition
 	 * resources. We walk the vnodes associated with the mount point
 	 * until we get the needed worklist requests that we can reap.
 	 */
 	if ((resource == FLUSH_BLOCKS_WAIT && 
 	     fs->fs_cstotal.cs_nbfree <= needed) ||
 	    (resource == FLUSH_INODES_WAIT && fs->fs_pendinginodes > 0 &&
 	     fs->fs_cstotal.cs_nifree <= needed)) {
 		MNT_VNODE_FOREACH_ALL(lvp, mp, mvp) {
 			if (TAILQ_FIRST(&lvp->v_bufobj.bo_dirty.bv_hd) == 0) {
 				VI_UNLOCK(lvp);
 				continue;
 			}
 			if (vget(lvp, LK_EXCLUSIVE | LK_INTERLOCK | LK_NOWAIT,
 			    curthread))
 				continue;
 			if (lvp->v_vflag & VV_NOSYNC) {	/* unlinked */
 				vput(lvp);
 				continue;
 			}
 			(void) ffs_syncvnode(lvp, MNT_NOWAIT, 0);
 			vput(lvp);
 		}
 		lvp = ump->um_devvp;
 		if (vn_lock(lvp, LK_EXCLUSIVE | LK_NOWAIT) == 0) {
 			VOP_FSYNC(lvp, MNT_NOWAIT, curthread);
 			VOP_UNLOCK(lvp, 0);
 		}
 		if (ump->softdep_on_worklist > 0) {
 			stat_cleanup_retries += 1;
 			goto retry;
 		}
 		stat_cleanup_failures += 1;
 	}
 	if (time_second - starttime > stat_cleanup_high_delay)
 		stat_cleanup_high_delay = time_second - starttime;
 	UFS_LOCK(ump);
 	return (1);
 }
 
 /*
  * If memory utilization has gotten too high, deliberately slow things
  * down and speed up the I/O processing.
  */
 static int
 request_cleanup(mp, resource)
 	struct mount *mp;
 	int resource;
 {
 	struct thread *td = curthread;
 	struct ufsmount *ump;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 	/*
 	 * We never hold up the filesystem syncer or buf daemon.
 	 */
 	if (td->td_pflags & (TDP_SOFTDEP|TDP_NORUNNINGBUF))
 		return (0);
 	/*
 	 * First check to see if the work list has gotten backlogged.
 	 * If it has, co-opt this process to help clean up two entries.
 	 * Because this process may hold inodes locked, we cannot
 	 * handle any remove requests that might block on a locked
 	 * inode as that could lead to deadlock.  We set TDP_SOFTDEP
 	 * to avoid recursively processing the worklist.
 	 */
 	if (ump->softdep_on_worklist > max_softdeps / 10) {
 		td->td_pflags |= TDP_SOFTDEP;
 		process_worklist_item(mp, 2, LK_NOWAIT);
 		td->td_pflags &= ~TDP_SOFTDEP;
 		stat_worklist_push += 2;
 		return(1);
 	}
 	/*
 	 * Next, we attempt to speed up the syncer process. If that
 	 * is successful, then we allow the process to continue.
 	 */
 	if (softdep_speedup(ump) &&
 	    resource != FLUSH_BLOCKS_WAIT &&
 	    resource != FLUSH_INODES_WAIT)
 		return(0);
 	/*
 	 * If we are resource constrained on inode dependencies, try
 	 * flushing some dirty inodes. Otherwise, we are constrained
 	 * by file deletions, so try accelerating flushes of directories
 	 * with removal dependencies. We would like to do the cleanup
 	 * here, but we probably hold an inode locked at this point and 
 	 * that might deadlock against one that we try to clean. So,
 	 * the best that we can do is request the syncer daemon to do
 	 * the cleanup for us.
 	 */
 	switch (resource) {
 
 	case FLUSH_INODES:
 	case FLUSH_INODES_WAIT:
 		ACQUIRE_GBLLOCK(&lk);
 		stat_ino_limit_push += 1;
 		req_clear_inodedeps += 1;
 		FREE_GBLLOCK(&lk);
 		stat_countp = &stat_ino_limit_hit;
 		break;
 
 	case FLUSH_BLOCKS:
 	case FLUSH_BLOCKS_WAIT:
 		ACQUIRE_GBLLOCK(&lk);
 		stat_blk_limit_push += 1;
 		req_clear_remove += 1;
 		FREE_GBLLOCK(&lk);
 		stat_countp = &stat_blk_limit_hit;
 		break;
 
 	default:
 		panic("request_cleanup: unknown type");
 	}
 	/*
 	 * Hopefully the syncer daemon will catch up and awaken us.
 	 * We wait at most tickdelay before proceeding in any case.
 	 */
 	ACQUIRE_GBLLOCK(&lk);
 	FREE_LOCK(ump);
 	proc_waiting += 1;
 	if (callout_pending(&softdep_callout) == FALSE)
 		callout_reset(&softdep_callout, tickdelay > 2 ? tickdelay : 2,
 		    pause_timer, 0);
 
 	msleep((caddr_t)&proc_waiting, &lk, PPAUSE, "softupdate", 0);
 	proc_waiting -= 1;
 	FREE_GBLLOCK(&lk);
 	ACQUIRE_LOCK(ump);
 	return (1);
 }
 
 /*
  * Awaken processes pausing in request_cleanup and clear proc_waiting
  * to indicate that there is no longer a timer running. Pause_timer
  * will be called with the global softdep mutex (&lk) locked.
  */
 static void
 pause_timer(arg)
 	void *arg;
 {
 
 	GBLLOCK_OWNED(&lk);
 	/*
 	 * The callout_ API has acquired mtx and will hold it around this
 	 * function call.
 	 */
 	*stat_countp += proc_waiting;
 	wakeup(&proc_waiting);
 }
 
 /*
  * If requested, try removing inode or removal dependencies.
  */
 static void
 check_clear_deps(mp)
 	struct mount *mp;
 {
 
 	/*
 	 * If we are suspended, it may be because of our using
 	 * too many inodedeps, so help clear them out.
 	 */
 	if (MOUNTEDSUJ(mp) && VFSTOUFS(mp)->softdep_jblocks->jb_suspended)
 		clear_inodedeps(mp);
 	/*
 	 * General requests for cleanup of backed up dependencies
 	 */
 	ACQUIRE_GBLLOCK(&lk);
 	if (req_clear_inodedeps) {
 		req_clear_inodedeps -= 1;
 		FREE_GBLLOCK(&lk);
 		clear_inodedeps(mp);
 		ACQUIRE_GBLLOCK(&lk);
 		wakeup(&proc_waiting);
 	}
 	if (req_clear_remove) {
 		req_clear_remove -= 1;
 		FREE_GBLLOCK(&lk);
 		clear_remove(mp);
 		ACQUIRE_GBLLOCK(&lk);
 		wakeup(&proc_waiting);
 	}
 	FREE_GBLLOCK(&lk);
 }
 
 /*
  * Flush out a directory with at least one removal dependency in an effort to
  * reduce the number of dirrem, freefile, and freeblks dependency structures.
  */
 static void
 clear_remove(mp)
 	struct mount *mp;
 {
 	struct pagedep_hashhead *pagedephd;
 	struct pagedep *pagedep;
 	struct ufsmount *ump;
 	struct vnode *vp;
 	struct bufobj *bo;
 	int error, cnt;
 	ino_t ino;
 
 	ump = VFSTOUFS(mp);
 	LOCK_OWNED(ump);
 
 	for (cnt = 0; cnt <= ump->pagedep_hash_size; cnt++) {
 		pagedephd = &ump->pagedep_hashtbl[ump->pagedep_nextclean++];
 		if (ump->pagedep_nextclean > ump->pagedep_hash_size)
 			ump->pagedep_nextclean = 0;
 		LIST_FOREACH(pagedep, pagedephd, pd_hash) {
 			if (LIST_EMPTY(&pagedep->pd_dirremhd))
 				continue;
 			ino = pagedep->pd_ino;
 			if (vn_start_write(NULL, &mp, V_NOWAIT) != 0)
 				continue;
 			FREE_LOCK(ump);
 
 			/*
 			 * Let unmount clear deps
 			 */
 			error = vfs_busy(mp, MBF_NOWAIT);
 			if (error != 0)
 				goto finish_write;
 			error = ffs_vgetf(mp, ino, LK_EXCLUSIVE, &vp,
 			     FFSV_FORCEINSMQ);
 			vfs_unbusy(mp);
 			if (error != 0) {
 				softdep_error("clear_remove: vget", error);
 				goto finish_write;
 			}
 			if ((error = ffs_syncvnode(vp, MNT_NOWAIT, 0)))
 				softdep_error("clear_remove: fsync", error);
 			bo = &vp->v_bufobj;
 			BO_LOCK(bo);
 			drain_output(vp);
 			BO_UNLOCK(bo);
 			vput(vp);
 		finish_write:
 			vn_finished_write(mp);
 			ACQUIRE_LOCK(ump);
 			return;
 		}
 	}
 }
 
 /*
  * Clear out a block of dirty inodes in an effort to reduce
  * the number of inodedep dependency structures.
  */
 static void
 clear_inodedeps(mp)
 	struct mount *mp;
 {
 	struct inodedep_hashhead *inodedephd;
 	struct inodedep *inodedep;
 	struct ufsmount *ump;
 	struct vnode *vp;
 	struct fs *fs;
 	int error, cnt;
 	ino_t firstino, lastino, ino;
 
 	ump = VFSTOUFS(mp);
 	fs = ump->um_fs;
 	LOCK_OWNED(ump);
 	/*
 	 * Pick a random inode dependency to be cleared.
 	 * We will then gather up all the inodes in its block 
 	 * that have dependencies and flush them out.
 	 */
 	for (cnt = 0; cnt <= ump->inodedep_hash_size; cnt++) {
 		inodedephd = &ump->inodedep_hashtbl[ump->inodedep_nextclean++];
 		if (ump->inodedep_nextclean > ump->inodedep_hash_size)
 			ump->inodedep_nextclean = 0;
 		if ((inodedep = LIST_FIRST(inodedephd)) != NULL)
 			break;
 	}
 	if (inodedep == NULL)
 		return;
 	/*
 	 * Find the last inode in the block with dependencies.
 	 */
 	firstino = inodedep->id_ino & ~(INOPB(fs) - 1);
 	for (lastino = firstino + INOPB(fs) - 1; lastino > firstino; lastino--)
 		if (inodedep_lookup(mp, lastino, 0, &inodedep) != 0)
 			break;
 	/*
 	 * Asynchronously push all but the last inode with dependencies.
 	 * Synchronously push the last inode with dependencies to ensure
 	 * that the inode block gets written to free up the inodedeps.
 	 */
 	for (ino = firstino; ino <= lastino; ino++) {
 		if (inodedep_lookup(mp, ino, 0, &inodedep) == 0)
 			continue;
 		if (vn_start_write(NULL, &mp, V_NOWAIT) != 0)
 			continue;
 		FREE_LOCK(ump);
 		error = vfs_busy(mp, MBF_NOWAIT); /* Let unmount clear deps */
 		if (error != 0) {
 			vn_finished_write(mp);
 			ACQUIRE_LOCK(ump);
 			return;
 		}
 		if ((error = ffs_vgetf(mp, ino, LK_EXCLUSIVE, &vp,
 		    FFSV_FORCEINSMQ)) != 0) {
 			softdep_error("clear_inodedeps: vget", error);
 			vfs_unbusy(mp);
 			vn_finished_write(mp);
 			ACQUIRE_LOCK(ump);
 			return;
 		}
 		vfs_unbusy(mp);
 		if (ino == lastino) {
 			if ((error = ffs_syncvnode(vp, MNT_WAIT, 0)))
 				softdep_error("clear_inodedeps: fsync1", error);
 		} else {
 			if ((error = ffs_syncvnode(vp, MNT_NOWAIT, 0)))
 				softdep_error("clear_inodedeps: fsync2", error);
 			BO_LOCK(&vp->v_bufobj);
 			drain_output(vp);
 			BO_UNLOCK(&vp->v_bufobj);
 		}
 		vput(vp);
 		vn_finished_write(mp);
 		ACQUIRE_LOCK(ump);
 	}
 }
 
 void
 softdep_buf_append(bp, wkhd)
 	struct buf *bp;
 	struct workhead *wkhd;
 {
 	struct worklist *wk;
 	struct ufsmount *ump;
 
 	if ((wk = LIST_FIRST(wkhd)) == NULL)
 		return;
 	KASSERT(MOUNTEDSOFTDEP(wk->wk_mp) != 0,
 	    ("softdep_buf_append called on non-softdep filesystem"));
 	ump = VFSTOUFS(wk->wk_mp);
 	ACQUIRE_LOCK(ump);
 	while ((wk = LIST_FIRST(wkhd)) != NULL) {
 		WORKLIST_REMOVE(wk);
 		WORKLIST_INSERT(&bp->b_dep, wk);
 	}
 	FREE_LOCK(ump);
 
 }
 
 void
 softdep_inode_append(ip, cred, wkhd)
 	struct inode *ip;
 	struct ucred *cred;
 	struct workhead *wkhd;
 {
 	struct buf *bp;
 	struct fs *fs;
 	int error;
 
 	KASSERT(MOUNTEDSOFTDEP(UFSTOVFS(ip->i_ump)) != 0,
 	    ("softdep_inode_append called on non-softdep filesystem"));
 	fs = ip->i_fs;
 	error = bread(ip->i_devvp, fsbtodb(fs, ino_to_fsba(fs, ip->i_number)),
 	    (int)fs->fs_bsize, cred, &bp);
 	if (error) {
 		bqrelse(bp);
 		softdep_freework(wkhd);
 		return;
 	}
 	softdep_buf_append(bp, wkhd);
 	bqrelse(bp);
 }
 
 void
 softdep_freework(wkhd)
 	struct workhead *wkhd;
 {
 	struct worklist *wk;
 	struct ufsmount *ump;
 
 	if ((wk = LIST_FIRST(wkhd)) == NULL)
 		return;
 	KASSERT(MOUNTEDSOFTDEP(wk->wk_mp) != 0,
 	    ("softdep_freework called on non-softdep filesystem"));
 	ump = VFSTOUFS(wk->wk_mp);
 	ACQUIRE_LOCK(ump);
 	handle_jwork(wkhd);
 	FREE_LOCK(ump);
 }
 
 /*
  * Function to determine if the buffer has outstanding dependencies
  * that will cause a roll-back if the buffer is written. If wantcount
  * is set, return number of dependencies, otherwise just yes or no.
  */
 static int
 softdep_count_dependencies(bp, wantcount)
 	struct buf *bp;
 	int wantcount;
 {
 	struct worklist *wk;
 	struct ufsmount *ump;
 	struct bmsafemap *bmsafemap;
 	struct freework *freework;
 	struct inodedep *inodedep;
 	struct indirdep *indirdep;
 	struct freeblks *freeblks;
 	struct allocindir *aip;
 	struct pagedep *pagedep;
 	struct dirrem *dirrem;
 	struct newblk *newblk;
 	struct mkdir *mkdir;
 	struct diradd *dap;
 	int i, retval;
 
 	retval = 0;
 	if ((wk = LIST_FIRST(&bp->b_dep)) == NULL)
 		return (0);
 	ump = VFSTOUFS(wk->wk_mp);
 	ACQUIRE_LOCK(ump);
 	LIST_FOREACH(wk, &bp->b_dep, wk_list) {
 		switch (wk->wk_type) {
 
 		case D_INODEDEP:
 			inodedep = WK_INODEDEP(wk);
 			if ((inodedep->id_state & DEPCOMPLETE) == 0) {
 				/* bitmap allocation dependency */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			if (TAILQ_FIRST(&inodedep->id_inoupdt)) {
 				/* direct block pointer dependency */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			if (TAILQ_FIRST(&inodedep->id_extupdt)) {
 				/* direct block pointer dependency */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			if (TAILQ_FIRST(&inodedep->id_inoreflst)) {
 				/* Add reference dependency. */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			continue;
 
 		case D_INDIRDEP:
 			indirdep = WK_INDIRDEP(wk);
 
 			TAILQ_FOREACH(freework, &indirdep->ir_trunc, fw_next) {
 				/* indirect truncation dependency */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 
 			LIST_FOREACH(aip, &indirdep->ir_deplisthd, ai_next) {
 				/* indirect block pointer dependency */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			continue;
 
 		case D_PAGEDEP:
 			pagedep = WK_PAGEDEP(wk);
 			LIST_FOREACH(dirrem, &pagedep->pd_dirremhd, dm_next) {
 				if (LIST_FIRST(&dirrem->dm_jremrefhd)) {
 					/* Journal remove ref dependency. */
 					retval += 1;
 					if (!wantcount)
 						goto out;
 				}
 			}
 			for (i = 0; i < DAHASHSZ; i++) {
 
 				LIST_FOREACH(dap, &pagedep->pd_diraddhd[i], da_pdlist) {
 					/* directory entry dependency */
 					retval += 1;
 					if (!wantcount)
 						goto out;
 				}
 			}
 			continue;
 
 		case D_BMSAFEMAP:
 			bmsafemap = WK_BMSAFEMAP(wk);
 			if (LIST_FIRST(&bmsafemap->sm_jaddrefhd)) {
 				/* Add reference dependency. */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			if (LIST_FIRST(&bmsafemap->sm_jnewblkhd)) {
 				/* Allocate block dependency. */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			continue;
 
 		case D_FREEBLKS:
 			freeblks = WK_FREEBLKS(wk);
 			if (LIST_FIRST(&freeblks->fb_jblkdephd)) {
 				/* Freeblk journal dependency. */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			continue;
 
 		case D_ALLOCDIRECT:
 		case D_ALLOCINDIR:
 			newblk = WK_NEWBLK(wk);
 			if (newblk->nb_jnewblk) {
 				/* Journal allocate dependency. */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			continue;
 
 		case D_MKDIR:
 			mkdir = WK_MKDIR(wk);
 			if (mkdir->md_jaddref) {
 				/* Journal reference dependency. */
 				retval += 1;
 				if (!wantcount)
 					goto out;
 			}
 			continue;
 
 		case D_FREEWORK:
 		case D_FREEDEP:
 		case D_JSEGDEP:
 		case D_JSEG:
 		case D_SBDEP:
 			/* never a dependency on these blocks */
 			continue;
 
 		default:
 			panic("softdep_count_dependencies: Unexpected type %s",
 			    TYPENAME(wk->wk_type));
 			/* NOTREACHED */
 		}
 	}
 out:
 	FREE_LOCK(ump);
 	return retval;
 }
 
 /*
  * Acquire exclusive access to a buffer.
  * Must be called with a locked mtx parameter.
  * Return acquired buffer or NULL on failure.
  */
 static struct buf *
 getdirtybuf(bp, lock, waitfor)
 	struct buf *bp;
 	struct rwlock *lock;
 	int waitfor;
 {
 	int error;
 
 	if (BUF_LOCK(bp, LK_EXCLUSIVE | LK_NOWAIT, NULL) != 0) {
 		if (waitfor != MNT_WAIT)
 			return (NULL);
 		error = BUF_LOCK(bp,
 		    LK_EXCLUSIVE | LK_SLEEPFAIL | LK_INTERLOCK, lock);
 		/*
 		 * Even if we sucessfully acquire bp here, we have dropped
 		 * lock, which may violates our guarantee.
 		 */
 		if (error == 0)
 			BUF_UNLOCK(bp);
 		else if (error != ENOLCK)
 			panic("getdirtybuf: inconsistent lock: %d", error);
 		rw_wlock(lock);
 		return (NULL);
 	}
 	if ((bp->b_vflags & BV_BKGRDINPROG) != 0) {
 		if (lock != BO_LOCKPTR(bp->b_bufobj) && waitfor == MNT_WAIT) {
 			rw_wunlock(lock);
 			BO_LOCK(bp->b_bufobj);
 			BUF_UNLOCK(bp);
 			if ((bp->b_vflags & BV_BKGRDINPROG) != 0) {
 				bp->b_vflags |= BV_BKGRDWAIT;
 				msleep(&bp->b_xflags, BO_LOCKPTR(bp->b_bufobj),
 				       PRIBIO | PDROP, "getbuf", 0);
 			} else
 				BO_UNLOCK(bp->b_bufobj);
 			rw_wlock(lock);
 			return (NULL);
 		}
 		BUF_UNLOCK(bp);
 		if (waitfor != MNT_WAIT)
 			return (NULL);
 		/*
 		 * The lock argument must be bp->b_vp's mutex in
 		 * this case.
 		 */
 #ifdef	DEBUG_VFS_LOCKS
 		if (bp->b_vp->v_type != VCHR)
 			ASSERT_BO_WLOCKED(bp->b_bufobj);
 #endif
 		bp->b_vflags |= BV_BKGRDWAIT;
 		rw_sleep(&bp->b_xflags, lock, PRIBIO, "getbuf", 0);
 		return (NULL);
 	}
 	if ((bp->b_flags & B_DELWRI) == 0) {
 		BUF_UNLOCK(bp);
 		return (NULL);
 	}
 	bremfree(bp);
 	return (bp);
 }
 
 
 /*
  * Check if it is safe to suspend the file system now.  On entry,
  * the vnode interlock for devvp should be held.  Return 0 with
  * the mount interlock held if the file system can be suspended now,
  * otherwise return EAGAIN with the mount interlock held.
  */
 int
 softdep_check_suspend(struct mount *mp,
 		      struct vnode *devvp,
 		      int softdep_depcnt,
 		      int softdep_accdepcnt,
 		      int secondary_writes,
 		      int secondary_accwrites)
 {
 	struct bufobj *bo;
 	struct ufsmount *ump;
 	int error;
 
 	bo = &devvp->v_bufobj;
 	ASSERT_BO_WLOCKED(bo);
 
 	/*
 	 * If we are not running with soft updates, then we need only
 	 * deal with secondary writes as we try to suspend.
 	 */
 	if (MOUNTEDSOFTDEP(mp) == 0) {
 		MNT_ILOCK(mp);
 		while (mp->mnt_secondary_writes != 0) {
 			BO_UNLOCK(bo);
 			msleep(&mp->mnt_secondary_writes, MNT_MTX(mp),
 			    (PUSER - 1) | PDROP, "secwr", 0);
 			BO_LOCK(bo);
 			MNT_ILOCK(mp);
 		}
 
 		/*
 		 * Reasons for needing more work before suspend:
 		 * - Dirty buffers on devvp.
 		 * - Secondary writes occurred after start of vnode sync loop
 		 */
 		error = 0;
 		if (bo->bo_numoutput > 0 ||
 		    bo->bo_dirty.bv_cnt > 0 ||
 		    secondary_writes != 0 ||
 		    mp->mnt_secondary_writes != 0 ||
 		    secondary_accwrites != mp->mnt_secondary_accwrites)
 			error = EAGAIN;
 		BO_UNLOCK(bo);
 		return (error);
 	}
 
 	/*
 	 * If we are running with soft updates, then we need to coordinate
 	 * with them as we try to suspend.
 	 */
 	ump = VFSTOUFS(mp);
 	for (;;) {
 		if (!TRY_ACQUIRE_LOCK(ump)) {
 			BO_UNLOCK(bo);
 			ACQUIRE_LOCK(ump);
 			FREE_LOCK(ump);
 			BO_LOCK(bo);
 			continue;
 		}
 		MNT_ILOCK(mp);
 		if (mp->mnt_secondary_writes != 0) {
 			FREE_LOCK(ump);
 			BO_UNLOCK(bo);
 			msleep(&mp->mnt_secondary_writes,
 			       MNT_MTX(mp),
 			       (PUSER - 1) | PDROP, "secwr", 0);
 			BO_LOCK(bo);
 			continue;
 		}
 		break;
 	}
 
 	/*
 	 * Reasons for needing more work before suspend:
 	 * - Dirty buffers on devvp.
 	 * - Softdep activity occurred after start of vnode sync loop
 	 * - Secondary writes occurred after start of vnode sync loop
 	 */
 	error = 0;
 	if (bo->bo_numoutput > 0 ||
 	    bo->bo_dirty.bv_cnt > 0 ||
 	    softdep_depcnt != 0 ||
 	    ump->softdep_deps != 0 ||
 	    softdep_accdepcnt != ump->softdep_accdeps ||
 	    secondary_writes != 0 ||
 	    mp->mnt_secondary_writes != 0 ||
 	    secondary_accwrites != mp->mnt_secondary_accwrites)
 		error = EAGAIN;
 	FREE_LOCK(ump);
 	BO_UNLOCK(bo);
 	return (error);
 }
 
 
 /*
  * Get the number of dependency structures for the file system, both
  * the current number and the total number allocated.  These will
  * later be used to detect that softdep processing has occurred.
  */
 void
 softdep_get_depcounts(struct mount *mp,
 		      int *softdep_depsp,
 		      int *softdep_accdepsp)
 {
 	struct ufsmount *ump;
 
 	if (MOUNTEDSOFTDEP(mp) == 0) {
 		*softdep_depsp = 0;
 		*softdep_accdepsp = 0;
 		return;
 	}
 	ump = VFSTOUFS(mp);
 	ACQUIRE_LOCK(ump);
 	*softdep_depsp = ump->softdep_deps;
 	*softdep_accdepsp = ump->softdep_accdeps;
 	FREE_LOCK(ump);
 }
 
 /*
  * Wait for pending output on a vnode to complete.
  * Must be called with vnode lock and interlock locked.
  *
  * XXX: Should just be a call to bufobj_wwait().
  */
 static void
 drain_output(vp)
 	struct vnode *vp;
 {
 	struct bufobj *bo;
 
 	bo = &vp->v_bufobj;
 	ASSERT_VOP_LOCKED(vp, "drain_output");
 	ASSERT_BO_WLOCKED(bo);
 
 	while (bo->bo_numoutput) {
 		bo->bo_flag |= BO_WWAIT;
 		msleep((caddr_t)&bo->bo_numoutput,
 		    BO_LOCKPTR(bo), PRIBIO + 1, "drainvp", 0);
 	}
 }
 
 /*
  * Called whenever a buffer that is being invalidated or reallocated
  * contains dependencies. This should only happen if an I/O error has
  * occurred. The routine is called with the buffer locked.
  */ 
 static void
 softdep_deallocate_dependencies(bp)
 	struct buf *bp;
 {
 
 	if ((bp->b_ioflags & BIO_ERROR) == 0)
 		panic("softdep_deallocate_dependencies: dangling deps");
 	if (bp->b_vp != NULL && bp->b_vp->v_mount != NULL)
 		softdep_error(bp->b_vp->v_mount->mnt_stat.f_mntonname, bp->b_error);
 	else
 		printf("softdep_deallocate_dependencies: "
 		    "got error %d while accessing filesystem\n", bp->b_error);
 	if (bp->b_error != ENXIO)
 		panic("softdep_deallocate_dependencies: unrecovered I/O error");
 }
 
 /*
  * Function to handle asynchronous write errors in the filesystem.
  */
 static void
 softdep_error(func, error)
 	char *func;
 	int error;
 {
 
 	/* XXX should do something better! */
 	printf("%s: got error %d while accessing filesystem\n", func, error);
 }
 
 #ifdef DDB
 
 static void
 inodedep_print(struct inodedep *inodedep, int verbose)
 {
 	db_printf("%p fs %p st %x ino %jd inoblk %jd delta %d nlink %d"
 	    " saveino %p\n",
 	    inodedep, inodedep->id_fs, inodedep->id_state,
 	    (intmax_t)inodedep->id_ino,
 	    (intmax_t)fsbtodb(inodedep->id_fs,
 	    ino_to_fsba(inodedep->id_fs, inodedep->id_ino)),
 	    inodedep->id_nlinkdelta, inodedep->id_savednlink,
 	    inodedep->id_savedino1);
 
 	if (verbose == 0)
 		return;
 
 	db_printf("\tpendinghd %p, bufwait %p, inowait %p, inoreflst %p, "
 	    "mkdiradd %p\n",
 	    LIST_FIRST(&inodedep->id_pendinghd),
 	    LIST_FIRST(&inodedep->id_bufwait),
 	    LIST_FIRST(&inodedep->id_inowait),
 	    TAILQ_FIRST(&inodedep->id_inoreflst),
 	    inodedep->id_mkdiradd);
 	db_printf("\tinoupdt %p, newinoupdt %p, extupdt %p, newextupdt %p\n",
 	    TAILQ_FIRST(&inodedep->id_inoupdt),
 	    TAILQ_FIRST(&inodedep->id_newinoupdt),
 	    TAILQ_FIRST(&inodedep->id_extupdt),
 	    TAILQ_FIRST(&inodedep->id_newextupdt));
 }
 
 DB_SHOW_COMMAND(inodedep, db_show_inodedep)
 {
 
 	if (have_addr == 0) {
 		db_printf("Address required\n");
 		return;
 	}
 	inodedep_print((struct inodedep*)addr, 1);
 }
 
 DB_SHOW_COMMAND(inodedeps, db_show_inodedeps)
 {
 	struct inodedep_hashhead *inodedephd;
 	struct inodedep *inodedep;
 	struct ufsmount *ump;
 	int cnt;
 
 	if (have_addr == 0) {
 		db_printf("Address required\n");
 		return;
 	}
 	ump = (struct ufsmount *)addr;
 	for (cnt = 0; cnt < ump->inodedep_hash_size; cnt++) {
 		inodedephd = &ump->inodedep_hashtbl[cnt];
 		LIST_FOREACH(inodedep, inodedephd, id_hash) {
 			inodedep_print(inodedep, 0);
 		}
 	}
 }
 
 DB_SHOW_COMMAND(worklist, db_show_worklist)
 {
 	struct worklist *wk;
 
 	if (have_addr == 0) {
 		db_printf("Address required\n");
 		return;
 	}
 	wk = (struct worklist *)addr;
 	printf("worklist: %p type %s state 0x%X\n",
 	    wk, TYPENAME(wk->wk_type), wk->wk_state);
 }
 
 DB_SHOW_COMMAND(workhead, db_show_workhead)
 {
 	struct workhead *wkhd;
 	struct worklist *wk;
 	int i;
 
 	if (have_addr == 0) {
 		db_printf("Address required\n");
 		return;
 	}
 	wkhd = (struct workhead *)addr;
 	wk = LIST_FIRST(wkhd);
 	for (i = 0; i < 100 && wk != NULL; i++, wk = LIST_NEXT(wk, wk_list))
 		db_printf("worklist: %p type %s state 0x%X",
 		    wk, TYPENAME(wk->wk_type), wk->wk_state);
 	if (i == 100)
 		db_printf("workhead overflow");
 	printf("\n");
 }
 
 
 DB_SHOW_COMMAND(mkdirs, db_show_mkdirs)
 {
 	struct mkdirlist *mkdirlisthd;
 	struct jaddref *jaddref;
 	struct diradd *diradd;
 	struct mkdir *mkdir;
 
 	if (have_addr == 0) {
 		db_printf("Address required\n");
 		return;
 	}
 	mkdirlisthd = (struct mkdirlist *)addr;
 	LIST_FOREACH(mkdir, mkdirlisthd, md_mkdirs) {
 		diradd = mkdir->md_diradd;
 		db_printf("mkdir: %p state 0x%X dap %p state 0x%X",
 		    mkdir, mkdir->md_state, diradd, diradd->da_state);
 		if ((jaddref = mkdir->md_jaddref) != NULL)
 			db_printf(" jaddref %p jaddref state 0x%X",
 			    jaddref, jaddref->ja_state);
 		db_printf("\n");
 	}
 }
 
 /* exported to ffs_vfsops.c */
 extern void db_print_ffs(struct ufsmount *ump);
 void
 db_print_ffs(struct ufsmount *ump)
 {
 	db_printf("mp %p %s devvp %p fs %p su_wl %d su_deps %d su_req %d\n",
 	    ump->um_mountp, ump->um_mountp->mnt_stat.f_mntonname,
 	    ump->um_devvp, ump->um_fs, ump->softdep_on_worklist,
 	    ump->softdep_deps, ump->softdep_req);
 }
 
 #endif /* DDB */
 
 #endif /* SOFTUPDATES */
Index: projects/clang360-import/sys/ufs/ffs/softdep.h
===================================================================
--- projects/clang360-import/sys/ufs/ffs/softdep.h	(revision 277944)
+++ projects/clang360-import/sys/ufs/ffs/softdep.h	(revision 277945)
@@ -1,1096 +1,1098 @@
 /*-
  * Copyright 1998, 2000 Marshall Kirk McKusick. All Rights Reserved.
  *
  * The soft updates code is derived from the appendix of a University
  * of Michigan technical report (Gregory R. Ganger and Yale N. Patt,
  * "Soft Updates: A Solution to the Metadata Update Problem in File
  * Systems", CSE-TR-254-95, August 1995).
  *
  * Further information about soft updates can be obtained from:
  *
  *	Marshall Kirk McKusick		http://www.mckusick.com/softdep/
  *	1614 Oxford Street		mckusick@mckusick.com
  *	Berkeley, CA 94709-1608		+1-510-843-9542
  *	USA
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  *
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  *
  * THIS SOFTWARE IS PROVIDED BY MARSHALL KIRK MCKUSICK ``AS IS'' AND ANY
  * EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
  * WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE
  * DISCLAIMED.  IN NO EVENT SHALL MARSHALL KIRK MCKUSICK BE LIABLE FOR
  * ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  *
  *	@(#)softdep.h	9.7 (McKusick) 6/21/00
  * $FreeBSD$
  */
 
 #include <sys/queue.h>
 
 /*
  * Allocation dependencies are handled with undo/redo on the in-memory
  * copy of the data. A particular data dependency is eliminated when
  * it is ALLCOMPLETE: that is ATTACHED, DEPCOMPLETE, and COMPLETE.
  * 
  * The ATTACHED flag means that the data is not currently being written
  * to disk.
  * 
  * The UNDONE flag means that the data has been rolled back to a safe
  * state for writing to the disk. When the I/O completes, the data is
  * restored to its current form and the state reverts to ATTACHED.
  * The data must be locked throughout the rollback, I/O, and roll
  * forward so that the rolled back information is never visible to
  * user processes.
  *
  * The COMPLETE flag indicates that the item has been written. For example,
  * a dependency that requires that an inode be written will be marked
  * COMPLETE after the inode has been written to disk.
  * 
  * The DEPCOMPLETE flag indicates the completion of any other
  * dependencies such as the writing of a cylinder group map has been
  * completed. A dependency structure may be freed only when both it
  * and its dependencies have completed and any rollbacks that are in
  * progress have finished as indicated by the set of ALLCOMPLETE flags
  * all being set.
  * 
  * The two MKDIR flags indicate additional dependencies that must be done
  * when creating a new directory. MKDIR_BODY is cleared when the directory
  * data block containing the "." and ".." entries has been written.
  * MKDIR_PARENT is cleared when the parent inode with the increased link
  * count for ".." has been written. When both MKDIR flags have been
  * cleared, the DEPCOMPLETE flag is set to indicate that the directory
  * dependencies have been completed. The writing of the directory inode
  * itself sets the COMPLETE flag which then allows the directory entry for
  * the new directory to be written to disk. The RMDIR flag marks a dirrem
  * structure as representing the removal of a directory rather than a
  * file. When the removal dependencies are completed, additional work needs
  * to be done* (an additional decrement of the associated inode, and a
  * decrement of the parent inode).
  *
  * The DIRCHG flag marks a diradd structure as representing the changing
  * of an existing entry rather than the addition of a new one. When
  * the update is complete the dirrem associated with the inode for
  * the old name must be added to the worklist to do the necessary
  * reference count decrement.
  * 
  * The GOINGAWAY flag indicates that the data structure is frozen from
  * further change until its dependencies have been completed and its
  * resources freed after which it will be discarded.
  *
  * The IOSTARTED flag prevents multiple calls to the I/O start routine from
  * doing multiple rollbacks.
  *
  * The NEWBLOCK flag marks pagedep structures that have just been allocated,
  * so must be claimed by the inode before all dependencies are complete.
  *
  * The INPROGRESS flag marks worklist structures that are still on the
  * worklist, but are being considered for action by some process.
  *
  * The UFS1FMT flag indicates that the inode being processed is a ufs1 format.
  *
  * The EXTDATA flag indicates that the allocdirect describes an
  * extended-attributes dependency.
  *
  * The ONWORKLIST flag shows whether the structure is currently linked
  * onto a worklist.
  *
  * The UNLINK* flags track the progress of updating the on-disk linked
  * list of active but unlinked inodes. When an inode is first unlinked
  * it is marked as UNLINKED. When its on-disk di_freelink has been
  * written its UNLINKNEXT flags is set. When its predecessor in the
  * list has its di_freelink pointing at us its UNLINKPREV is set.
  * When the on-disk list can reach it from the superblock, its
  * UNLINKONLIST flag is set. Once all of these flags are set, it
  * is safe to let its last name be removed.
  */
 #define	ATTACHED	0x000001
 #define	UNDONE		0x000002
 #define	COMPLETE	0x000004
 #define	DEPCOMPLETE	0x000008
 #define	MKDIR_PARENT	0x000010 /* diradd, mkdir, jaddref, jsegdep only */
 #define	MKDIR_BODY	0x000020 /* diradd, mkdir, jaddref only */
 #define	RMDIR		0x000040 /* dirrem only */
 #define	DIRCHG		0x000080 /* diradd, dirrem only */
 #define	GOINGAWAY	0x000100 /* indirdep, jremref only */
 #define	IOSTARTED	0x000200 /* inodedep, pagedep, bmsafemap only */
 #define	DELAYEDFREE	0x000400 /* allocindirect free delayed. */
 #define	NEWBLOCK	0x000800 /* pagedep, jaddref only */
 #define	INPROGRESS	0x001000 /* dirrem, freeblks, freefrag, freefile only */
 #define	UFS1FMT		0x002000 /* indirdep only */
 #define	EXTDATA		0x004000 /* allocdirect only */
 #define	ONWORKLIST	0x008000
 #define	IOWAITING	0x010000 /* Thread is waiting for IO to complete. */
 #define	ONDEPLIST	0x020000 /* Structure is on a dependency list. */
 #define	UNLINKED	0x040000 /* inodedep has been unlinked. */
 #define	UNLINKNEXT	0x080000 /* inodedep has valid di_freelink */
 #define	UNLINKPREV	0x100000 /* inodedep is pointed at in the unlink list */
 #define	UNLINKONLIST	0x200000 /* inodedep is in the unlinked list on disk */
 #define	UNLINKLINKS	(UNLINKNEXT | UNLINKPREV)
 
 #define	ALLCOMPLETE	(ATTACHED | COMPLETE | DEPCOMPLETE)
 
 /*
  * Values for each of the soft dependency types.
  */
 #define	D_PAGEDEP	0
 #define	D_INODEDEP	1
 #define	D_BMSAFEMAP	2
 #define	D_NEWBLK	3
 #define	D_ALLOCDIRECT	4
 #define	D_INDIRDEP	5
 #define	D_ALLOCINDIR	6
 #define	D_FREEFRAG	7
 #define	D_FREEBLKS	8
 #define	D_FREEFILE	9
 #define	D_DIRADD	10
 #define	D_MKDIR		11
 #define	D_DIRREM	12
 #define	D_NEWDIRBLK	13
 #define	D_FREEWORK	14
 #define	D_FREEDEP	15
 #define	D_JADDREF	16
 #define	D_JREMREF	17
 #define	D_JMVREF	18
 #define	D_JNEWBLK	19
 #define	D_JFREEBLK	20
 #define	D_JFREEFRAG	21
 #define	D_JSEG		22
 #define	D_JSEGDEP	23
 #define	D_SBDEP		24
 #define	D_JTRUNC	25
 #define	D_JFSYNC	26
 #define	D_SENTINEL	27
 #define	D_LAST		D_SENTINEL
 
 /*
  * The workitem queue.
  * 
  * It is sometimes useful and/or necessary to clean up certain dependencies
  * in the background rather than during execution of an application process
  * or interrupt service routine. To realize this, we append dependency
  * structures corresponding to such tasks to a "workitem" queue. In a soft
  * updates implementation, most pending workitems should not wait for more
  * than a couple of seconds, so the filesystem syncer process awakens once
  * per second to process the items on the queue.
  */
 
 /* LIST_HEAD(workhead, worklist);	-- declared in buf.h */
 
 /*
  * Each request can be linked onto a work queue through its worklist structure.
  * To avoid the need for a pointer to the structure itself, this structure
  * MUST be declared FIRST in each type in which it appears! If more than one
  * worklist is needed in the structure, then a wk_data field must be added
  * and the macros below changed to use it.
  */
 struct worklist {
 	LIST_ENTRY(worklist)	wk_list;	/* list of work requests */
 	struct mount		*wk_mp;		/* Mount we live in */
 	unsigned int		wk_type:8,	/* type of request */
 				wk_state:24;	/* state flags */
 };
 #define	WK_DATA(wk) ((void *)(wk))
 #define	WK_PAGEDEP(wk) ((struct pagedep *)(wk))
 #define	WK_INODEDEP(wk) ((struct inodedep *)(wk))
 #define	WK_BMSAFEMAP(wk) ((struct bmsafemap *)(wk))
 #define	WK_NEWBLK(wk)  ((struct newblk *)(wk))
 #define	WK_ALLOCDIRECT(wk) ((struct allocdirect *)(wk))
 #define	WK_INDIRDEP(wk) ((struct indirdep *)(wk))
 #define	WK_ALLOCINDIR(wk) ((struct allocindir *)(wk))
 #define	WK_FREEFRAG(wk) ((struct freefrag *)(wk))
 #define	WK_FREEBLKS(wk) ((struct freeblks *)(wk))
 #define	WK_FREEWORK(wk) ((struct freework *)(wk))
 #define	WK_FREEFILE(wk) ((struct freefile *)(wk))
 #define	WK_DIRADD(wk) ((struct diradd *)(wk))
 #define	WK_MKDIR(wk) ((struct mkdir *)(wk))
 #define	WK_DIRREM(wk) ((struct dirrem *)(wk))
 #define	WK_NEWDIRBLK(wk) ((struct newdirblk *)(wk))
 #define	WK_JADDREF(wk) ((struct jaddref *)(wk))
 #define	WK_JREMREF(wk) ((struct jremref *)(wk))
 #define	WK_JMVREF(wk) ((struct jmvref *)(wk))
 #define	WK_JSEGDEP(wk) ((struct jsegdep *)(wk))
 #define	WK_JSEG(wk) ((struct jseg *)(wk))
 #define	WK_JNEWBLK(wk) ((struct jnewblk *)(wk))
 #define	WK_JFREEBLK(wk) ((struct jfreeblk *)(wk))
 #define	WK_FREEDEP(wk) ((struct freedep *)(wk))
 #define	WK_JFREEFRAG(wk) ((struct jfreefrag *)(wk))
 #define	WK_SBDEP(wk) ((struct sbdep *)(wk))
 #define	WK_JTRUNC(wk) ((struct jtrunc *)(wk))
 #define	WK_JFSYNC(wk) ((struct jfsync *)(wk))
 
 /*
  * Various types of lists
  */
 LIST_HEAD(dirremhd, dirrem);
 LIST_HEAD(diraddhd, diradd);
 LIST_HEAD(newblkhd, newblk);
 LIST_HEAD(inodedephd, inodedep);
 LIST_HEAD(allocindirhd, allocindir);
 LIST_HEAD(allocdirecthd, allocdirect);
 TAILQ_HEAD(allocdirectlst, allocdirect);
 LIST_HEAD(indirdephd, indirdep);
 LIST_HEAD(jaddrefhd, jaddref);
 LIST_HEAD(jremrefhd, jremref);
 LIST_HEAD(jmvrefhd, jmvref);
 LIST_HEAD(jnewblkhd, jnewblk);
 LIST_HEAD(jblkdephd, jblkdep);
 LIST_HEAD(freeworkhd, freework);
 TAILQ_HEAD(freeworklst, freework);
 TAILQ_HEAD(jseglst, jseg);
 TAILQ_HEAD(inoreflst, inoref);
 TAILQ_HEAD(freeblklst, freeblks);
 
 /*
  * The "pagedep" structure tracks the various dependencies related to
  * a particular directory page. If a directory page has any dependencies,
  * it will have a pagedep linked to its associated buffer. The
  * pd_dirremhd list holds the list of dirrem requests which decrement
  * inode reference counts. These requests are processed after the
  * directory page with the corresponding zero'ed entries has been
  * written. The pd_diraddhd list maintains the list of diradd requests
  * which cannot be committed until their corresponding inode has been
  * written to disk. Because a directory may have many new entries
  * being created, several lists are maintained hashed on bits of the
  * offset of the entry into the directory page to keep the lists from
  * getting too long. Once a new directory entry has been cleared to
  * be written, it is moved to the pd_pendinghd list. After the new
  * entry has been written to disk it is removed from the pd_pendinghd
  * list, any removed operations are done, and the dependency structure
  * is freed.
  */
 #define	DAHASHSZ 5
 #define	DIRADDHASH(offset) (((offset) >> 2) % DAHASHSZ)
 struct pagedep {
 	struct	worklist pd_list;	/* page buffer */
 #	define	pd_state pd_list.wk_state /* check for multiple I/O starts */
 	LIST_ENTRY(pagedep) pd_hash;	/* hashed lookup */
 	ino_t	pd_ino;			/* associated file */
 	ufs_lbn_t pd_lbn;		/* block within file */
 	struct	newdirblk *pd_newdirblk; /* associated newdirblk if NEWBLOCK */
 	struct	dirremhd pd_dirremhd;	/* dirrem's waiting for page */
 	struct	diraddhd pd_diraddhd[DAHASHSZ]; /* diradd dir entry updates */
 	struct	diraddhd pd_pendinghd;	/* directory entries awaiting write */
 	struct	jmvrefhd pd_jmvrefhd;	/* Dependent journal writes. */
 };
 
 /*
  * The "inodedep" structure tracks the set of dependencies associated
  * with an inode. One task that it must manage is delayed operations
  * (i.e., work requests that must be held until the inodedep's associated
  * inode has been written to disk). Getting an inode from its incore 
  * state to the disk requires two steps to be taken by the filesystem
  * in this order: first the inode must be copied to its disk buffer by
  * the VOP_UPDATE operation; second the inode's buffer must be written
  * to disk. To ensure that both operations have happened in the required
  * order, the inodedep maintains two lists. Delayed operations are
  * placed on the id_inowait list. When the VOP_UPDATE is done, all
  * operations on the id_inowait list are moved to the id_bufwait list.
  * When the buffer is written, the items on the id_bufwait list can be
  * safely moved to the work queue to be processed. A second task of the
  * inodedep structure is to track the status of block allocation within
  * the inode.  Each block that is allocated is represented by an
  * "allocdirect" structure (see below). It is linked onto the id_newinoupdt
  * list until both its contents and its allocation in the cylinder
  * group map have been written to disk. Once these dependencies have been
  * satisfied, it is removed from the id_newinoupdt list and any followup
  * actions such as releasing the previous block or fragment are placed
  * on the id_inowait list. When an inode is updated (a VOP_UPDATE is
  * done), the "inodedep" structure is linked onto the buffer through
  * its worklist. Thus, it will be notified when the buffer is about
  * to be written and when it is done. At the update time, all the
  * elements on the id_newinoupdt list are moved to the id_inoupdt list
  * since those changes are now relevant to the copy of the inode in the
  * buffer. Also at update time, the tasks on the id_inowait list are
  * moved to the id_bufwait list so that they will be executed when
  * the updated inode has been written to disk. When the buffer containing
  * the inode is written to disk, any updates listed on the id_inoupdt
  * list are rolled back as they are not yet safe. Following the write,
  * the changes are once again rolled forward and any actions on the
  * id_bufwait list are processed (since those actions are now safe).
  * The entries on the id_inoupdt and id_newinoupdt lists must be kept
  * sorted by logical block number to speed the calculation of the size
  * of the rolled back inode (see explanation in initiate_write_inodeblock).
  * When a directory entry is created, it is represented by a diradd.
  * The diradd is added to the id_inowait list as it cannot be safely
  * written to disk until the inode that it represents is on disk. After
  * the inode is written, the id_bufwait list is processed and the diradd
  * entries are moved to the id_pendinghd list where they remain until
  * the directory block containing the name has been written to disk.
  * The purpose of keeping the entries on the id_pendinghd list is so that
  * the softdep_fsync function can find and push the inode's directory
  * name(s) as part of the fsync operation for that file.
  */
 struct inodedep {
 	struct	worklist id_list;	/* buffer holding inode block */
 #	define	id_state id_list.wk_state /* inode dependency state */
 	LIST_ENTRY(inodedep) id_hash;	/* hashed lookup */
 	TAILQ_ENTRY(inodedep) id_unlinked;	/* Unlinked but ref'd inodes */
 	struct	fs *id_fs;		/* associated filesystem */
 	ino_t	id_ino;			/* dependent inode */
 	nlink_t	id_nlinkdelta;		/* saved effective link count */
 	nlink_t	id_savednlink;		/* Link saved during rollback */
 	LIST_ENTRY(inodedep) id_deps;	/* bmsafemap's list of inodedep's */
 	struct	bmsafemap *id_bmsafemap; /* related bmsafemap (if pending) */
 	struct	diradd *id_mkdiradd;	/* diradd for a mkdir. */
 	struct	inoreflst id_inoreflst;	/* Inode reference adjustments. */
 	long	id_savedextsize;	/* ext size saved during rollback */
 	off_t	id_savedsize;		/* file size saved during rollback */
 	struct	dirremhd id_dirremhd;	/* Removals pending. */
 	struct	workhead id_pendinghd;	/* entries awaiting directory write */
 	struct	workhead id_bufwait;	/* operations after inode written */
 	struct	workhead id_inowait;	/* operations waiting inode update */
 	struct	allocdirectlst id_inoupdt; /* updates before inode written */
 	struct	allocdirectlst id_newinoupdt; /* updates when inode written */
 	struct	allocdirectlst id_extupdt; /* extdata updates pre-inode write */
 	struct	allocdirectlst id_newextupdt; /* extdata updates at ino write */
 	struct	freeblklst id_freeblklst; /* List of partial truncates. */
 	union {
 	struct	ufs1_dinode *idu_savedino1; /* saved ufs1_dinode contents */
 	struct	ufs2_dinode *idu_savedino2; /* saved ufs2_dinode contents */
 	} id_un;
 };
 #define	id_savedino1 id_un.idu_savedino1
 #define	id_savedino2 id_un.idu_savedino2
 
 /*
  * A "bmsafemap" structure maintains a list of dependency structures
  * that depend on the update of a particular cylinder group map.
  * It has lists for newblks, allocdirects, allocindirs, and inodedeps.
  * It is attached to the buffer of a cylinder group block when any of
  * these things are allocated from the cylinder group. It is freed
  * after the cylinder group map is written and the state of its
  * dependencies are updated with DEPCOMPLETE to indicate that it has
  * been processed.
  */
 struct bmsafemap {
 	struct	worklist sm_list;	/* cylgrp buffer */
 #	define	sm_state sm_list.wk_state
 	LIST_ENTRY(bmsafemap) sm_hash;	/* Hash links. */
 	LIST_ENTRY(bmsafemap) sm_next;	/* Mount list. */
 	int	sm_cg;
 	struct	buf *sm_buf;		/* associated buffer */
 	struct	allocdirecthd sm_allocdirecthd; /* allocdirect deps */
 	struct	allocdirecthd sm_allocdirectwr; /* writing allocdirect deps */
 	struct	allocindirhd sm_allocindirhd; /* allocindir deps */
 	struct	allocindirhd sm_allocindirwr; /* writing allocindir deps */
 	struct	inodedephd sm_inodedephd; /* inodedep deps */
 	struct	inodedephd sm_inodedepwr; /* writing inodedep deps */
 	struct	newblkhd sm_newblkhd;	/* newblk deps */
 	struct	newblkhd sm_newblkwr;	/* writing newblk deps */
 	struct	jaddrefhd sm_jaddrefhd;	/* Pending inode allocations. */
 	struct	jnewblkhd sm_jnewblkhd;	/* Pending block allocations. */
 	struct	workhead sm_freehd;	/* Freedep deps. */
 	struct	workhead sm_freewr;	/* Written freedeps. */
 };
 
 /*
  * A "newblk" structure is attached to a bmsafemap structure when a block
  * or fragment is allocated from a cylinder group. Its state is set to
  * DEPCOMPLETE when its cylinder group map is written. It is converted to
  * an allocdirect or allocindir allocation once the allocator calls the
  * appropriate setup function. It will initially be linked onto a bmsafemap
  * list. Once converted it can be linked onto the lists described for
  * allocdirect or allocindir as described below.
  */ 
 struct newblk {
 	struct	worklist nb_list;	/* See comment above. */
 #	define	nb_state nb_list.wk_state
 	LIST_ENTRY(newblk) nb_hash;	/* Hashed lookup. */
 	LIST_ENTRY(newblk) nb_deps;	/* Bmsafemap's list of newblks. */
 	struct	jnewblk *nb_jnewblk;	/* New block journal entry. */
 	struct	bmsafemap *nb_bmsafemap;/* Cylgrp dep (if pending). */
 	struct	freefrag *nb_freefrag;	/* Fragment to be freed (if any). */
 	struct	indirdephd nb_indirdeps; /* Children indirect blocks. */
 	struct	workhead nb_newdirblk;	/* Dir block to notify when written. */
 	struct	workhead nb_jwork;	/* Journal work pending. */
 	ufs2_daddr_t	nb_newblkno;	/* New value of block pointer. */
 };
 
 /*
  * An "allocdirect" structure is attached to an "inodedep" when a new block
  * or fragment is allocated and pointed to by the inode described by
  * "inodedep". The worklist is linked to the buffer that holds the block.
  * When the block is first allocated, it is linked to the bmsafemap
  * structure associated with the buffer holding the cylinder group map
  * from which it was allocated. When the cylinder group map is written
  * to disk, ad_state has the DEPCOMPLETE flag set. When the block itself
  * is written, the COMPLETE flag is set. Once both the cylinder group map
  * and the data itself have been written, it is safe to write the inode
  * that claims the block. If there was a previous fragment that had been
  * allocated before the file was increased in size, the old fragment may
  * be freed once the inode claiming the new block is written to disk.
  * This ad_fragfree request is attached to the id_inowait list of the
  * associated inodedep (pointed to by ad_inodedep) for processing after
  * the inode is written. When a block is allocated to a directory, an
  * fsync of a file whose name is within that block must ensure not only
  * that the block containing the file name has been written, but also
  * that the on-disk inode references that block. When a new directory
  * block is created, we allocate a newdirblk structure which is linked
  * to the associated allocdirect (on its ad_newdirblk list). When the
  * allocdirect has been satisfied, the newdirblk structure is moved to
  * the inodedep id_bufwait list of its directory to await the inode
  * being written. When the inode is written, the directory entries are
  * fully committed and can be deleted from their pagedep->id_pendinghd
  * and inodedep->id_pendinghd lists.
  */
 struct allocdirect {
 	struct	newblk ad_block;	/* Common block logic */
 #	define	ad_list ad_block.nb_list /* block pointer worklist */
 #	define	ad_state ad_list.wk_state /* block pointer state */
 	TAILQ_ENTRY(allocdirect) ad_next; /* inodedep's list of allocdirect's */
 	struct	inodedep *ad_inodedep;	/* associated inodedep */
 	ufs2_daddr_t	ad_oldblkno;	/* old value of block pointer */
 	int		ad_offset;	/* Pointer offset in parent. */
 	long		ad_newsize;	/* size of new block */
 	long		ad_oldsize;	/* size of old block */
 };
 #define	ad_newblkno	ad_block.nb_newblkno
 #define	ad_freefrag	ad_block.nb_freefrag
 #define	ad_newdirblk	ad_block.nb_newdirblk
 
 /*
  * A single "indirdep" structure manages all allocation dependencies for
  * pointers in an indirect block. The up-to-date state of the indirect
  * block is stored in ir_savedata. The set of pointers that may be safely
  * written to the disk is stored in ir_safecopy. The state field is used
  * only to track whether the buffer is currently being written (in which
  * case it is not safe to update ir_safecopy). Ir_deplisthd contains the
  * list of allocindir structures, one for each block that needs to be
  * written to disk. Once the block and its bitmap allocation have been
  * written the safecopy can be updated to reflect the allocation and the
  * allocindir structure freed. If ir_state indicates that an I/O on the
  * indirect block is in progress when ir_safecopy is to be updated, the
  * update is deferred by placing the allocindir on the ir_donehd list.
  * When the I/O on the indirect block completes, the entries on the
  * ir_donehd list are processed by updating their corresponding ir_safecopy
  * pointers and then freeing the allocindir structure.
  */
 struct indirdep {
 	struct	worklist ir_list;	/* buffer holding indirect block */
 #	define	ir_state ir_list.wk_state /* indirect block pointer state */
 	LIST_ENTRY(indirdep) ir_next;	/* alloc{direct,indir} list */
 	TAILQ_HEAD(, freework) ir_trunc;	/* List of truncations. */
 	caddr_t	ir_saveddata;		/* buffer cache contents */
 	struct	buf *ir_savebp;		/* buffer holding safe copy */
 	struct	buf *ir_bp;		/* buffer holding live copy */
 	struct	allocindirhd ir_completehd; /* waiting for indirdep complete */
 	struct	allocindirhd ir_writehd; /* Waiting for the pointer write. */
 	struct	allocindirhd ir_donehd;	/* done waiting to update safecopy */
 	struct	allocindirhd ir_deplisthd; /* allocindir deps for this block */
 	struct	freeblks *ir_freeblks;	/* Freeblks that frees this indir. */
 };
 
 /*
  * An "allocindir" structure is attached to an "indirdep" when a new block
  * is allocated and pointed to by the indirect block described by the
  * "indirdep". The worklist is linked to the buffer that holds the new block.
  * When the block is first allocated, it is linked to the bmsafemap
  * structure associated with the buffer holding the cylinder group map
  * from which it was allocated. When the cylinder group map is written
  * to disk, ai_state has the DEPCOMPLETE flag set. When the block itself
  * is written, the COMPLETE flag is set. Once both the cylinder group map
  * and the data itself have been written, it is safe to write the entry in
  * the indirect block that claims the block; the "allocindir" dependency 
  * can then be freed as it is no longer applicable.
  */
 struct allocindir {
 	struct	newblk ai_block;	/* Common block area */
 #	define	ai_state ai_block.nb_list.wk_state /* indirect pointer state */
 	LIST_ENTRY(allocindir) ai_next;	/* indirdep's list of allocindir's */
 	struct	indirdep *ai_indirdep;	/* address of associated indirdep */
 	ufs2_daddr_t	ai_oldblkno;	/* old value of block pointer */
 	ufs_lbn_t	ai_lbn;		/* Logical block number. */
 	int		ai_offset;	/* Pointer offset in parent. */
 };
 #define	ai_newblkno	ai_block.nb_newblkno
 #define	ai_freefrag	ai_block.nb_freefrag
 #define	ai_newdirblk	ai_block.nb_newdirblk
 
 /*
  * The allblk union is used to size the newblk structure on allocation so
  * that it may be any one of three types.
  */
 union allblk {
 	struct	allocindir ab_allocindir;
 	struct	allocdirect ab_allocdirect;
 	struct	newblk	ab_newblk;
 };
 
 /*
  * A "freefrag" structure is attached to an "inodedep" when a previously
  * allocated fragment is replaced with a larger fragment, rather than extended.
  * The "freefrag" structure is constructed and attached when the replacement
  * block is first allocated. It is processed after the inode claiming the
  * bigger block that replaces it has been written to disk.
  */
 struct freefrag {
 	struct	worklist ff_list;	/* id_inowait or delayed worklist */
 #	define	ff_state ff_list.wk_state
 	struct	worklist *ff_jdep;	/* Associated journal entry. */
 	struct	workhead ff_jwork;	/* Journal work pending. */
 	ufs2_daddr_t ff_blkno;		/* fragment physical block number */
 	long	ff_fragsize;		/* size of fragment being deleted */
 	ino_t	ff_inum;		/* owning inode number */
 	enum	vtype ff_vtype;		/* owning inode's file type */
 };
 
 /*
  * A "freeblks" structure is attached to an "inodedep" when the
  * corresponding file's length is reduced to zero. It records all
  * the information needed to free the blocks of a file after its
  * zero'ed inode has been written to disk.  The actual work is done
  * by child freework structures which are responsible for individual
  * inode pointers while freeblks is responsible for retiring the
  * entire operation when it is complete and holding common members.
  */
 struct freeblks {
 	struct	worklist fb_list;	/* id_inowait or delayed worklist */
 #	define	fb_state fb_list.wk_state /* inode and dirty block state */
 	TAILQ_ENTRY(freeblks) fb_next;	/* List of inode truncates. */
 	struct	jblkdephd fb_jblkdephd;	/* Journal entries pending */
 	struct	workhead fb_freeworkhd;	/* Work items pending */
 	struct	workhead fb_jwork;	/* Journal work pending */
 	struct	vnode *fb_devvp;	/* filesystem device vnode */
 #ifdef QUOTA
 	struct	dquot *fb_quota[MAXQUOTAS]; /* quotas to be adjusted */
 #endif
 	uint64_t fb_modrev;		/* Inode revision at start of trunc. */
 	off_t	fb_len;			/* Length we're truncating to. */
 	ufs2_daddr_t fb_chkcnt;		/* Blocks released. */
 	ino_t	fb_inum;		/* inode owner of blocks */
 	enum	vtype fb_vtype;		/* inode owner's file type */
 	uid_t	fb_uid;			/* uid of previous owner of blocks */
 	int	fb_ref;			/* Children outstanding. */
 	int	fb_cgwait;		/* cg writes outstanding. */
 };
 
 /*
  * A "freework" structure handles the release of a tree of blocks or a single
  * block.  Each indirect block in a tree is allocated its own freework
  * structure so that the indirect block may be freed only when all of its
  * children are freed.  In this way we enforce the rule that an allocated
  * block must have a valid path to a root that is journaled.  Each child
  * block acquires a reference and when the ref hits zero the parent ref
  * is decremented.  If there is no parent the freeblks ref is decremented.
  */
 struct freework {
 	struct	worklist fw_list;		/* Delayed worklist. */
 #	define	fw_state fw_list.wk_state
 	LIST_ENTRY(freework) fw_segs;		/* Seg list. */
 	TAILQ_ENTRY(freework) fw_next;		/* Hash/Trunc list. */
 	struct	jnewblk	 *fw_jnewblk;		/* Journal entry to cancel. */
 	struct	freeblks *fw_freeblks;		/* Root of operation. */
 	struct	freework *fw_parent;		/* Parent indirect. */
 	struct	indirdep *fw_indir;		/* indirect block. */
 	ufs2_daddr_t	 fw_blkno;		/* Our block #. */
 	ufs_lbn_t	 fw_lbn;		/* Original lbn before free. */
 	uint16_t	 fw_frags;		/* Number of frags. */
 	uint16_t	 fw_ref;		/* Number of children out. */
 	uint16_t	 fw_off;		/* Current working position. */
 	uint16_t	 fw_start;		/* Start of partial truncate. */
 };
 
 /*
  * A "freedep" structure is allocated to track the completion of a bitmap
  * write for a freework.  One freedep may cover many freed blocks so long
  * as they reside in the same cylinder group.  When the cg is written
  * the freedep decrements the ref on the freework which may permit it
  * to be freed as well.
  */
 struct freedep {
 	struct	worklist fd_list;	/* Delayed worklist. */
 	struct	freework *fd_freework;	/* Parent freework. */
 };
 
 /*
  * A "freefile" structure is attached to an inode when its
  * link count is reduced to zero. It marks the inode as free in
  * the cylinder group map after the zero'ed inode has been written
  * to disk and any associated blocks and fragments have been freed.
  */
 struct freefile {
 	struct	worklist fx_list;	/* id_inowait or delayed worklist */
 	mode_t	fx_mode;		/* mode of inode */
 	ino_t	fx_oldinum;		/* inum of the unlinked file */
 	struct	vnode *fx_devvp;	/* filesystem device vnode */
 	struct	workhead fx_jwork;	/* journal work pending. */
 };
 
 /*
  * A "diradd" structure is linked to an "inodedep" id_inowait list when a
  * new directory entry is allocated that references the inode described
  * by "inodedep". When the inode itself is written (either the initial
  * allocation for new inodes or with the increased link count for
  * existing inodes), the COMPLETE flag is set in da_state. If the entry
  * is for a newly allocated inode, the "inodedep" structure is associated
  * with a bmsafemap which prevents the inode from being written to disk
  * until the cylinder group has been updated. Thus the da_state COMPLETE
  * flag cannot be set until the inode bitmap dependency has been removed.
  * When creating a new file, it is safe to write the directory entry that
  * claims the inode once the referenced inode has been written. Since
  * writing the inode clears the bitmap dependencies, the DEPCOMPLETE flag
  * in the diradd can be set unconditionally when creating a file. When
  * creating a directory, there are two additional dependencies described by
  * mkdir structures (see their description below). When these dependencies
  * are resolved the DEPCOMPLETE flag is set in the diradd structure.
  * If there are multiple links created to the same inode, there will be
  * a separate diradd structure created for each link. The diradd is
  * linked onto the pg_diraddhd list of the pagedep for the directory
  * page that contains the entry. When a directory page is written,
  * the pg_diraddhd list is traversed to rollback any entries that are
  * not yet ready to be written to disk. If a directory entry is being
  * changed (by rename) rather than added, the DIRCHG flag is set and
  * the da_previous entry points to the entry that will be "removed"
  * once the new entry has been committed. During rollback, entries
  * with da_previous are replaced with the previous inode number rather
  * than zero.
  *
  * The overlaying of da_pagedep and da_previous is done to keep the
  * structure down. If a da_previous entry is present, the pointer to its
  * pagedep is available in the associated dirrem entry. If the DIRCHG flag
  * is set, the da_previous entry is valid; if not set the da_pagedep entry
  * is valid. The DIRCHG flag never changes; it is set when the structure
  * is created if appropriate and is never cleared.
  */
 struct diradd {
 	struct	worklist da_list;	/* id_inowait or id_pendinghd list */
 #	define	da_state da_list.wk_state /* state of the new directory entry */
 	LIST_ENTRY(diradd) da_pdlist;	/* pagedep holding directory block */
 	doff_t	da_offset;		/* offset of new dir entry in dir blk */
 	ino_t	da_newinum;		/* inode number for the new dir entry */
 	union {
 	struct	dirrem *dau_previous;	/* entry being replaced in dir change */
 	struct	pagedep *dau_pagedep;	/* pagedep dependency for addition */
 	} da_un;
 	struct workhead da_jwork;	/* Journal work awaiting completion. */
 };
 #define	da_previous da_un.dau_previous
 #define	da_pagedep da_un.dau_pagedep
 
 /*
  * Two "mkdir" structures are needed to track the additional dependencies
  * associated with creating a new directory entry. Normally a directory
  * addition can be committed as soon as the newly referenced inode has been
  * written to disk with its increased link count. When a directory is
  * created there are two additional dependencies: writing the directory
  * data block containing the "." and ".." entries (MKDIR_BODY) and writing
  * the parent inode with the increased link count for ".." (MKDIR_PARENT).
  * These additional dependencies are tracked by two mkdir structures that
  * reference the associated "diradd" structure. When they have completed,
  * they set the DEPCOMPLETE flag on the diradd so that it knows that its
  * extra dependencies have been completed. The md_state field is used only
  * to identify which type of dependency the mkdir structure is tracking.
  * It is not used in the mainline code for any purpose other than consistency
  * checking. All the mkdir structures in the system are linked together on
  * a list. This list is needed so that a diradd can find its associated
  * mkdir structures and deallocate them if it is prematurely freed (as for
  * example if a mkdir is immediately followed by a rmdir of the same directory).
  * Here, the free of the diradd must traverse the list to find the associated
  * mkdir structures that reference it. The deletion would be faster if the
  * diradd structure were simply augmented to have two pointers that referenced
  * the associated mkdir's. However, this would increase the size of the diradd
  * structure to speed a very infrequent operation.
  */
 struct mkdir {
 	struct	worklist md_list;	/* id_inowait or buffer holding dir */
 #	define	md_state md_list.wk_state /* type: MKDIR_PARENT or MKDIR_BODY */
 	struct	diradd *md_diradd;	/* associated diradd */
 	struct	jaddref *md_jaddref;	/* dependent jaddref. */
 	struct	buf *md_buf;		/* MKDIR_BODY: buffer holding dir */
 	LIST_ENTRY(mkdir) md_mkdirs;	/* list of all mkdirs */
 };
 
 /*
  * A "dirrem" structure describes an operation to decrement the link
  * count on an inode. The dirrem structure is attached to the pg_dirremhd
  * list of the pagedep for the directory page that contains the entry.
  * It is processed after the directory page with the deleted entry has
  * been written to disk.
  */
 struct dirrem {
 	struct	worklist dm_list;	/* delayed worklist */
 #	define	dm_state dm_list.wk_state /* state of the old directory entry */
 	LIST_ENTRY(dirrem) dm_next;	/* pagedep's list of dirrem's */
 	LIST_ENTRY(dirrem) dm_inonext;	/* inodedep's list of dirrem's */
 	struct	jremrefhd dm_jremrefhd;	/* Pending remove reference deps. */
 	ino_t	dm_oldinum;		/* inum of the removed dir entry */
 	doff_t	dm_offset;		/* offset of removed dir entry in blk */
 	union {
 	struct	pagedep *dmu_pagedep;	/* pagedep dependency for remove */
 	ino_t	dmu_dirinum;		/* parent inode number (for rmdir) */
 	} dm_un;
 	struct workhead dm_jwork;	/* Journal work awaiting completion. */
 };
 #define	dm_pagedep dm_un.dmu_pagedep
 #define	dm_dirinum dm_un.dmu_dirinum
 
 /*
  * A "newdirblk" structure tracks the progress of a newly allocated
  * directory block from its creation until it is claimed by its on-disk
  * inode. When a block is allocated to a directory, an fsync of a file
  * whose name is within that block must ensure not only that the block
  * containing the file name has been written, but also that the on-disk
  * inode references that block. When a new directory block is created,
  * we allocate a newdirblk structure which is linked to the associated
  * allocdirect (on its ad_newdirblk list). When the allocdirect has been
  * satisfied, the newdirblk structure is moved to the inodedep id_bufwait
  * list of its directory to await the inode being written. When the inode
  * is written, the directory entries are fully committed and can be
  * deleted from their pagedep->id_pendinghd and inodedep->id_pendinghd
  * lists. Note that we could track directory blocks allocated to indirect
  * blocks using a similar scheme with the allocindir structures. Rather
  * than adding this level of complexity, we simply write those newly 
  * allocated indirect blocks synchronously as such allocations are rare.
  * In the case of a new directory the . and .. links are tracked with
  * a mkdir rather than a pagedep.  In this case we track the mkdir
  * so it can be released when it is written.  A workhead is used
  * to simplify canceling a mkdir that is removed by a subsequent dirrem.
  */
 struct newdirblk {
 	struct	worklist db_list;	/* id_inowait or pg_newdirblk */
 #	define	db_state db_list.wk_state
 	struct	pagedep *db_pagedep;	/* associated pagedep */
 	struct	workhead db_mkdir;
 };
 
 /*
  * The inoref structure holds the elements common to jaddref and jremref
  * so they may easily be queued in-order on the inodedep.
  */
 struct inoref {
 	struct	worklist if_list;	/* Journal pending or jseg entries. */
 #	define	if_state if_list.wk_state
 	TAILQ_ENTRY(inoref) if_deps;	/* Links for inodedep. */
 	struct	jsegdep	*if_jsegdep;	/* Will track our journal record. */
 	off_t		if_diroff;	/* Directory offset. */
 	ino_t		if_ino;		/* Inode number. */
 	ino_t		if_parent;	/* Parent inode number. */
 	nlink_t		if_nlink;	/* nlink before addition. */
 	uint16_t	if_mode;	/* File mode, needed for IFMT. */
 };
 
 /*
  * A "jaddref" structure tracks a new reference (link count) on an inode
  * and prevents the link count increase and bitmap allocation until a
  * journal entry can be written.  Once the journal entry is written,
  * the inode is put on the pendinghd of the bmsafemap and a diradd or
  * mkdir entry is placed on the bufwait list of the inode.  The DEPCOMPLETE
  * flag is used to indicate that all of the required information for writing
  * the journal entry is present.  MKDIR_BODY and MKDIR_PARENT are used to
  * differentiate . and .. links from regular file names.  NEWBLOCK indicates
  * a bitmap is still pending.  If a new reference is canceled by a delete
  * prior to writing the journal the jaddref write is canceled and the
  * structure persists to prevent any disk-visible changes until it is
  * ultimately released when the file is freed or the link is dropped again.
  */
 struct jaddref {
 	struct	inoref	ja_ref;		/* see inoref above. */
 #	define	ja_list	ja_ref.if_list	/* Jrnl pending, id_inowait, dm_jwork.*/
 #	define	ja_state ja_ref.if_list.wk_state
 	LIST_ENTRY(jaddref) ja_bmdeps;	/* Links for bmsafemap. */
 	union {
 		struct	diradd	*jau_diradd;	/* Pending diradd. */
 		struct	mkdir	*jau_mkdir;	/* MKDIR_{PARENT,BODY} */
 	} ja_un;
 };
 #define	ja_diradd	ja_un.jau_diradd
 #define	ja_mkdir	ja_un.jau_mkdir
 #define	ja_diroff	ja_ref.if_diroff
 #define	ja_ino		ja_ref.if_ino
 #define	ja_parent	ja_ref.if_parent
 #define	ja_mode		ja_ref.if_mode
 
 /*
  * A "jremref" structure tracks a removed reference (unlink) on an
  * inode and prevents the directory remove from proceeding until the
  * journal entry is written.  Once the journal has been written the remove
  * may proceed as normal. 
  */
 struct jremref {
 	struct	inoref	jr_ref;		/* see inoref above. */
 #	define	jr_list	jr_ref.if_list	/* Linked to softdep_journal_pending. */
 #	define	jr_state jr_ref.if_list.wk_state
 	LIST_ENTRY(jremref) jr_deps;	/* Links for dirrem. */
 	struct	dirrem	*jr_dirrem;	/* Back pointer to dirrem. */
 };
 
 /*
  * A "jmvref" structure tracks a name relocations within the same
  * directory block that occur as a result of directory compaction.
  * It prevents the updated directory entry from being written to disk
  * until the journal entry is written. Once the journal has been
  * written the compacted directory may be written to disk.
  */
 struct jmvref {
 	struct	worklist jm_list;	/* Linked to softdep_journal_pending. */
 	LIST_ENTRY(jmvref) jm_deps;	/* Jmvref on pagedep. */
 	struct pagedep	*jm_pagedep;	/* Back pointer to pagedep. */
 	ino_t		jm_parent;	/* Containing directory inode number. */
 	ino_t		jm_ino;		/* Inode number of our entry. */
 	off_t		jm_oldoff;	/* Our old offset in directory. */
 	off_t		jm_newoff;	/* Our new offset in directory. */
 };
 
 /*
  * A "jnewblk" structure tracks a newly allocated block or fragment and
  * prevents the direct or indirect block pointer as well as the cg bitmap
  * from being written until it is logged.  After it is logged the jsegdep
  * is attached to the allocdirect or allocindir until the operation is
  * completed or reverted.  If the operation is reverted prior to the journal
  * write the jnewblk structure is maintained to prevent the bitmaps from
  * reaching the disk.  Ultimately the jnewblk structure will be passed
  * to the free routine as the in memory cg is modified back to the free
  * state at which time it can be released. It may be held on any of the
  * fx_jwork, fw_jwork, fb_jwork, ff_jwork, nb_jwork, or ir_jwork lists.
  */
 struct jnewblk {
 	struct	worklist jn_list;	/* See lists above. */
 #	define	jn_state jn_list.wk_state
 	struct	jsegdep	*jn_jsegdep;	/* Will track our journal record. */
 	LIST_ENTRY(jnewblk) jn_deps;	/* Jnewblks on sm_jnewblkhd. */
 	struct	worklist *jn_dep;	/* Dependency to ref completed seg. */
 	ufs_lbn_t	jn_lbn;		/* Lbn to which allocated. */
 	ufs2_daddr_t	jn_blkno;	/* Blkno allocated */
 	ino_t		jn_ino;		/* Ino to which allocated. */
 	int		jn_oldfrags;	/* Previous fragments when extended. */
 	int		jn_frags;	/* Number of fragments. */
 };
 
 /*
  * A "jblkdep" structure tracks jfreeblk and jtrunc records attached to a
  * freeblks structure.
  */
 struct jblkdep {
 	struct	worklist jb_list;	/* For softdep journal pending. */
 	struct	jsegdep *jb_jsegdep;	/* Reference to the jseg. */
 	struct	freeblks *jb_freeblks;	/* Back pointer to freeblks. */
 	LIST_ENTRY(jblkdep) jb_deps;	/* Dep list on freeblks. */
 
 };
 
 /*
  * A "jfreeblk" structure tracks the journal write for freeing a block
  * or tree of blocks.  The block pointer must not be cleared in the inode
  * or indirect prior to the jfreeblk being written to the journal.
  */
 struct jfreeblk {
 	struct	jblkdep	jf_dep;		/* freeblks linkage. */
 	ufs_lbn_t	jf_lbn;		/* Lbn from which blocks freed. */
 	ufs2_daddr_t	jf_blkno;	/* Blkno being freed. */
 	ino_t		jf_ino;		/* Ino from which blocks freed. */
 	int		jf_frags;	/* Number of frags being freed. */
 };
 
 /*
  * A "jfreefrag" tracks the freeing of a single block when a fragment is
  * extended or an indirect page is replaced.  It is not part of a larger
  * freeblks operation.
  */
 struct jfreefrag {
 	struct	worklist fr_list;	/* Linked to softdep_journal_pending. */
 #	define	fr_state fr_list.wk_state
 	struct	jsegdep	*fr_jsegdep;	/* Will track our journal record. */
 	struct freefrag	*fr_freefrag;	/* Back pointer to freefrag. */
 	ufs_lbn_t	fr_lbn;		/* Lbn from which frag freed. */
 	ufs2_daddr_t	fr_blkno;	/* Blkno being freed. */
 	ino_t		fr_ino;		/* Ino from which frag freed. */
 	int		fr_frags;	/* Size of frag being freed. */
 };
 
 /*
  * A "jtrunc" journals the intent to truncate an inode's data or extent area.
  */
 struct jtrunc {
 	struct	jblkdep	jt_dep;		/* freeblks linkage. */
 	off_t		jt_size;	/* Final file size. */
 	int		jt_extsize;	/* Final extent size. */
 	ino_t		jt_ino;		/* Ino being truncated. */
 };
 
 /*
  * A "jfsync" journals the completion of an fsync which invalidates earlier
  * jtrunc records in the journal.
  */
 struct jfsync {
 	struct worklist	jfs_list;	/* For softdep journal pending. */
 	off_t		jfs_size;	/* Sync file size. */
 	int		jfs_extsize;	/* Sync extent size. */
 	ino_t		jfs_ino;	/* ino being synced. */
 };
 
 /*
  * A "jsegdep" structure tracks a single reference to a written journal
  * segment so the journal space can be reclaimed when all dependencies
  * have been written. It can hang off of id_inowait, dm_jwork, da_jwork,
  * nb_jwork, ff_jwork, or fb_jwork lists.
  */
 struct jsegdep {
 	struct	worklist jd_list;	/* See above for lists. */
 #	define	jd_state jd_list.wk_state
 	struct	jseg	*jd_seg;	/* Our journal record. */
 };
 
 /*
  * A "jseg" structure contains all of the journal records written in a
  * single disk write.  The jaddref and jremref structures are linked into
  * js_entries so thay may be completed when the write completes.  The
  * js_entries also include the write dependency structures: jmvref,
  * jnewblk, jfreeblk, jfreefrag, and jtrunc.  The js_refs field counts
  * the number of entries on the js_entries list. Thus there is a single
  * jseg entry to describe each journal write.
  */
 struct jseg {
 	struct	worklist js_list;	/* b_deps link for journal */
 #	define	js_state js_list.wk_state
 	struct	workhead js_entries;	/* Entries awaiting write */
 	LIST_HEAD(, freework) js_indirs;/* List of indirects in this seg. */
 	TAILQ_ENTRY(jseg) js_next;	/* List of all unfinished segments. */
 	struct	jblocks *js_jblocks;	/* Back pointer to block/seg list */
 	struct	buf *js_buf;		/* Buffer while unwritten */
 	uint64_t js_seq;		/* Journal record sequence number. */
 	uint64_t js_oldseq;		/* Oldest valid sequence number. */
 	int	js_size;		/* Size of journal record in bytes. */
 	int	js_cnt;			/* Total items allocated. */
 	int	js_refs;		/* Count of js_entries items. */
 };
 
 /*
  * A 'sbdep' structure tracks the head of the free inode list and
  * superblock writes.  This makes sure the superblock is always pointing at
  * the first possible unlinked inode for the suj recovery process.  If a
  * block write completes and we discover a new head is available the buf
  * is dirtied and the dep is kept. See the description of the UNLINK*
  * flags above for more details.
  */
 struct sbdep {
 	struct	worklist sb_list;	/* b_dep linkage */
 	struct	fs	*sb_fs;		/* Filesystem pointer within buf. */
 	struct	ufsmount *sb_ump;	/* Our mount structure */
 };
 
 /*
  * Private journaling structures.
  */
 struct jblocks {
 	struct jseglst	jb_segs;	/* TAILQ of current segments. */
 	struct jseg	*jb_writeseg;	/* Next write to complete. */
 	struct jseg	*jb_oldestseg;	/* Oldest segment with valid entries. */
 	struct jextent	*jb_extent;	/* Extent array. */
 	uint64_t	jb_nextseq;	/* Next sequence number. */
 	uint64_t	jb_oldestwrseq;	/* Oldest written sequence number. */
 	uint8_t		jb_needseg;	/* Need a forced segment. */
 	uint8_t		jb_suspended;	/* Did journal suspend writes? */
 	int		jb_avail;	/* Available extents. */
 	int		jb_used;	/* Last used extent. */
 	int		jb_head;	/* Allocator head. */
 	int		jb_off;		/* Allocator extent offset. */
 	int		jb_blocks;	/* Total disk blocks covered. */
 	int		jb_free;	/* Total disk blocks free. */
 	int		jb_min;		/* Minimum free space. */
 	int		jb_low;		/* Low on space. */
 	int		jb_age;		/* Insertion time of oldest rec. */
 };
 
 struct jextent {
 	ufs2_daddr_t	je_daddr;	/* Disk block address. */
 	int		je_blocks;	/* Disk block count. */
 };
 
 /*
  * Hash table declarations.
  */
 LIST_HEAD(mkdirlist, mkdir);
 LIST_HEAD(pagedep_hashhead, pagedep);
 LIST_HEAD(inodedep_hashhead, inodedep);
 LIST_HEAD(newblk_hashhead, newblk);
 LIST_HEAD(bmsafemap_hashhead, bmsafemap);
 TAILQ_HEAD(indir_hashhead, freework);
 
 /*
  * Per-filesystem soft dependency data.
  * Allocated at mount and freed at unmount.
  */
 struct mount_softdeps {
 	struct	rwlock sd_fslock;		/* softdep lock */
 	struct	workhead sd_workitem_pending;	/* softdep work queue */
 	struct	worklist *sd_worklist_tail;	/* Tail pointer for above */
 	struct	workhead sd_journal_pending;	/* journal work queue */
 	struct	worklist *sd_journal_tail;	/* Tail pointer for above */
 	struct	jblocks *sd_jblocks;		/* Journal block information */
 	struct	inodedeplst sd_unlinked;	/* Unlinked inodes */
 	struct	bmsafemaphd sd_dirtycg;		/* Dirty CGs */
 	struct	mkdirlist sd_mkdirlisthd;	/* Track mkdirs */
 	struct	pagedep_hashhead *sd_pdhash;	/* pagedep hash table */
 	u_long	sd_pdhashsize;			/* pagedep hash table size-1 */
 	long	sd_pdnextclean;			/* next hash bucket to clean */
 	struct	inodedep_hashhead *sd_idhash;	/* inodedep hash table */
 	u_long	sd_idhashsize;			/* inodedep hash table size-1 */
 	long	sd_idnextclean;			/* next hash bucket to clean */
 	struct	newblk_hashhead *sd_newblkhash;	/* newblk hash table */
 	u_long	sd_newblkhashsize;		/* newblk hash table size-1 */
 	struct	bmsafemap_hashhead *sd_bmhash;	/* bmsafemap hash table */
 	u_long	sd_bmhashsize;			/* bmsafemap hash table size-1*/
 	struct	indir_hashhead *sd_indirhash;	/* indir hash table */
 	u_long	sd_indirhashsize;		/* indir hash table size-1 */
 	int	sd_on_journal;			/* Items on the journal list */
 	int	sd_on_worklist;			/* Items on the worklist */
 	int	sd_deps;			/* Total dependency count */
 	int	sd_accdeps;			/* accumulated dep count */
 	int	sd_req;				/* Wakeup when deps hits 0. */
 	int	sd_flags;			/* comm with flushing thread */
 	int	sd_cleanups;			/* Calls to cleanup */
 	struct	thread *sd_flushtd;		/* thread handling flushing */
 	TAILQ_ENTRY(mount_softdeps) sd_next;	/* List of softdep filesystem */
 	struct	ufsmount *sd_ump;		/* our ufsmount structure */
 	u_long	sd_curdeps[D_LAST + 1];		/* count of current deps */
 };
 /*
  * Flags for communicating with the syncer thread.
  */
 #define FLUSH_EXIT	0x0001	/* time to exit */
 #define FLUSH_CLEANUP	0x0002	/* need to clear out softdep structures */
+#define	FLUSH_STARTING	0x0004	/* flush thread not yet started */
+
 /*
  * Keep the old names from when these were in the ufsmount structure.
  */
 #define	softdep_workitem_pending	um_softdep->sd_workitem_pending
 #define	softdep_worklist_tail		um_softdep->sd_worklist_tail
 #define	softdep_journal_pending		um_softdep->sd_journal_pending
 #define	softdep_journal_tail		um_softdep->sd_journal_tail
 #define	softdep_jblocks			um_softdep->sd_jblocks
 #define	softdep_unlinked		um_softdep->sd_unlinked
 #define	softdep_dirtycg			um_softdep->sd_dirtycg
 #define	softdep_mkdirlisthd		um_softdep->sd_mkdirlisthd
 #define	pagedep_hashtbl			um_softdep->sd_pdhash
 #define	pagedep_hash_size		um_softdep->sd_pdhashsize
 #define	pagedep_nextclean		um_softdep->sd_pdnextclean
 #define	inodedep_hashtbl		um_softdep->sd_idhash
 #define	inodedep_hash_size		um_softdep->sd_idhashsize
 #define	inodedep_nextclean		um_softdep->sd_idnextclean
 #define	newblk_hashtbl			um_softdep->sd_newblkhash
 #define	newblk_hash_size		um_softdep->sd_newblkhashsize
 #define	bmsafemap_hashtbl		um_softdep->sd_bmhash
 #define	bmsafemap_hash_size		um_softdep->sd_bmhashsize
 #define	indir_hashtbl			um_softdep->sd_indirhash
 #define	indir_hash_size			um_softdep->sd_indirhashsize
 #define	softdep_on_journal		um_softdep->sd_on_journal
 #define	softdep_on_worklist		um_softdep->sd_on_worklist
 #define	softdep_deps			um_softdep->sd_deps
 #define	softdep_accdeps			um_softdep->sd_accdeps
 #define	softdep_req			um_softdep->sd_req
 #define	softdep_flags			um_softdep->sd_flags
 #define	softdep_flushtd			um_softdep->sd_flushtd
 #define	softdep_curdeps			um_softdep->sd_curdeps
Index: projects/clang360-import/sys
===================================================================
--- projects/clang360-import/sys	(revision 277944)
+++ projects/clang360-import/sys	(revision 277945)

Property changes on: projects/clang360-import/sys
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head/sys:r277902-277944
Index: projects/clang360-import/tools/tools/nanobsd/rescue/build.sh
===================================================================
--- projects/clang360-import/tools/tools/nanobsd/rescue/build.sh	(revision 277944)
+++ projects/clang360-import/tools/tools/nanobsd/rescue/build.sh	(revision 277945)
@@ -1,42 +1,45 @@
 #!/bin/sh
 #
 # $FreeBSD$
 #
 
 today=`date '+%Y%m%d'`
 
 if [ -z "${1}" -o \! -f "${1}" ]; then
   echo "Usage: $0 cfg_file [-bhiknw]"
   echo "-i : skip image build"
   echo "-w : skip buildworld step"
   echo "-k : skip buildkernel step"
   echo "-b : skip buildworld and buildkernel step"
   exit
 fi
 
 CFG="${1}"
 shift;
 
 if [ \! -d /usr/obj/Rescue ]; then
   mkdir -p /usr/obj/Rescue
 fi
 
 sh ../nanobsd.sh $* -c ${CFG}
 
+if [ \! -d /usr/obj/Rescue ]; then
+  mkdir -p /usr/obj/Rescue
+fi
 F32="/usr/obj/Rescue/rescue_${today}_x32"
 D32="/usr/obj/nanobsd.rescue_i386"
 if [ -f "${D32}/_.disk.full" ]; then
-  mv "${D32}/_.disk.full" "${F32}.img"
+  cp "${D32}/_.disk.full" "${F32}.img"
 fi
 if [ -f "${D32}/_.disk.iso" ]; then
-  mv "${D32}/_.disk.iso" "${F32}.iso"
+  cp "${D32}/_.disk.iso" "${F32}.iso"
 fi
 
 F64="/usr/obj/Rescue/rescue_${today}_x64"
 D64="/usr/obj/nanobsd.rescue_amd64"
 if [ -f "${D64}/_.disk.full" ]; then
-  mv "${D64}/_.disk.full" "${F64}.img"
+  cp "${D64}/_.disk.full" "${F64}.img"
 fi
 if [ -f "${D64}/_.disk.iso" ]; then
-  mv "${D64}/_.disk.iso" "${F64}.iso"
+  cp "${D64}/_.disk.iso" "${F64}.iso"
 fi
Index: projects/clang360-import/tools/tools/nanobsd/rescue/common
===================================================================
--- projects/clang360-import/tools/tools/nanobsd/rescue/common	(revision 277944)
+++ projects/clang360-import/tools/tools/nanobsd/rescue/common	(revision 277945)
@@ -1,108 +1,124 @@
 #
 # $FreeBSD$
 #
 NANO_TOOLS=`pwd`
 NANO_PACKAGE_DIR=`pwd`/Pkg
 NANO_RAM_TMPVARSIZE=40960
 NANO_PMAKE="make -j 8"
 NANO_LABEL="rescue"
 NANO_RAM_TMPVARSIZE=40960
 #NANO_MEDIASIZE="8027712"
 #NANO_MEDIASIZE="2097152"
 NANO_MEDIASIZE="3932160"
 NANO_SECTS="63"
 NANO_HEADS="16"
 NANO_IMAGES="2"
 NANO_INIT_IMG2="0"
 NANO_BOOT0CFG="-o packet,update,nosetdrv -s 1 -m 3"
 NANO_DRIVE=da0
 #NANO_MODULES=
 NANO_BOOTLOADER="boot/boot0"
 NANO_BOOT2CFG=""
 NANO_MD_BACKING=swap
 
 # Options to put in make.conf during buildworld only
 CONF_BUILD='
 '
 # Options to put in make.conf during installworld only                          
 CONF_INSTALL='
 '
 # Options to put in make.conf during both build- & installworld.                
 CONF_WORLD='                                                                    
 #TARGET_ARCH=i386
 CFLAGS=-O -pipe                                                                
+WITHOUT_TESTS=YES
 ALL_MODULES=YES
 '
 
+# Functions
+toLower() {
+  echo $1 | tr "[:upper:]" "[:lower:]"
+}
+
+toUpper() {
+  echo $1 | tr "[:lower:]" "[:upper:]"
+}
+
 #customize_cmd cust_comconsole
 customize_cmd cust_allow_ssh_root
 customize_cmd cust_install_files
 
 cust_ld32_cfg () (
 	cd ${NANO_WORLDDIR}/libexec
 	if [ \! -f ld-elf32.so.1 ]; then
 	ln -s ld-elf.so.1 ld-elf32.so.1
 	fi
 )
 customize_cmd cust_ld32_cfg
 
 #cust_boot_cfg () (
 #	cd ${NANO_WORLDDIR}
 #	echo "-S115200 -h" > boot.config
 #	echo "console=\"comconsole\"" > boot/loader.conf
 #	echo "comconsole_speed=\"115200\"" >> boot/loader.conf
 #	echo "hint.acpi.0.disabled=\"1\"" >> boot/loader.conf
 #)
 #customize_cmd cust_boot_cfg
 
 customize_cmd cust_pkg
 
 cust_etc_cfg () (
   cd ${NANO_WORLDDIR}
 #  mkdir -pv scratch
 	echo "hostname=\"rescue\"" > etc/rc.conf
 	echo "font8x14=\"iso15-8x14\"" >> etc/rc.conf
 	echo "font8x16=\"iso15-8x16\"" >> etc/rc.conf
 	echo "font8x8=\"iso15-8x8\"" >> etc/rc.conf
 	echo "keymap=\"german.iso\"" >> etc/rc.conf
 	echo "#ifconfig_fxp0=\"AUTO\"" >> etc/rc.conf
 	echo "#sshd_enable=\"YES\"" >> etc/rc.conf
 	echo "/dev/ufs/${NANO_LABEL}s1a / ufs ro,noatime 0 0" > etc/fstab
 	echo "/dev/${NANO_DRIVE}s3 /cfg ufs rw,noauto 2 2" >> etc/fstab
 	echo "tmpfs /boot/zfs tmpfs rw,size=1048576,mode=777 0 0" >> etc/fstab
 	echo "ports:/usr/ports /usr/ports nfs rw,noauto,noatime,bg,soft,intr,nfsv3 0 0" >> etc/fstab
 #	echo "/dev/ad1s1a /scratch ufs rw,noauto,noatime 0 0" >> etc/fstab
 	/usr/sbin/pwd_mkdb -d etc etc/master.passwd
 )
 customize_cmd cust_etc_cfg
 
 setup_nanobsd_etc ( ) (
 	pprint 2 "configure nanobsd /etc"
 	(
 	cd ${NANO_WORLDDIR}
 	# create diskless marker file
 	touch etc/diskless
 	# Make root filesystem R/O by default
 	echo "root_rw_mount=NO" >> etc/defaults/rc.conf
 	# save config file for scripts
 	echo "NANO_DRIVE=${NANO_DRIVE}" > etc/nanobsd.conf
 	mkdir -p cfg
 	)
 )
 last_orders () (
 	pprint 2 "last orders"
 	(
 	cd ${NANO_WORLDDIR}
-	echo "/dev/iso9660/${NANO_LABEL} / cd9660 ro,noatime 0 0" > etc/fstab
+	#makefs converts labels to uppercase anyways
+	BIGLABEL=`toUpper "${NANO_LABEL}"`
+	echo "/dev/iso9660/${BIGLABEL} / cd9660 ro,noatime 0 0" > etc/fstab
 	echo "tmpfs /boot/zfs tmpfs rw,size=1048576,mode=777 0 0" >> etc/fstab
 	echo "ports:/usr/ports /usr/ports nfs rw,noauto,noatime,bg,soft,intr,nfsv3 0 0" >> etc/fstab
 #	echo "/dev/ad1s1a /scratch ufs rw,noauto,noatime 0 0" >> etc/fstab
 	rm -f conf/default/etc/remount
 	touch conf/default/etc/.keepme
 	touch conf/default/var/.keepme
+	mkdir bootpool
+	mkdir mnt/a
+	mkdir mnt/b
+	mkdir mnt/c
 	cd ..
 	makefs -t cd9660 -o rockridge \
-	-o label="${NANO_LABEL}" -o publisher="RMX" \
+	-o label="${BIGLABEL}" -o publisher="RMX" \
 	-o bootimage="i386;_.w/boot/cdboot" -o no-emul-boot _.disk.iso _.w/
 	)
 )
Index: projects/clang360-import/usr.bin/grep/Makefile
===================================================================
--- projects/clang360-import/usr.bin/grep/Makefile	(revision 277944)
+++ projects/clang360-import/usr.bin/grep/Makefile	(revision 277945)
@@ -1,88 +1,89 @@
 #	$NetBSD: Makefile,v 1.4 2011/02/16 01:31:33 joerg Exp $
 #	$FreeBSD$
 #	$OpenBSD: Makefile,v 1.6 2003/06/25 15:00:04 millert Exp $
 
 .include <src.opts.mk>
 
 .if ${MK_BSD_GREP} == "yes"
 PROG=	grep
 .else
 PROG=	bsdgrep
 CLEANFILES+= bsdgrep.1
 
 bsdgrep.1: grep.1
 	${CP} ${.ALLSRC} ${.TARGET}
 .endif
 SRCS=	file.c grep.c queue.c util.c
 
 # Extra files ported backported form some regex improvements
 .PATH: ${.CURDIR}/regex
 SRCS+=	fastmatch.c hashtable.c tre-compile.c tre-fastmatch.c xmalloc.c
 CFLAGS+=-I${.CURDIR}/regex
 
 .if ${MK_BSD_GREP} == "yes"
 LINKS=	${BINDIR}/grep ${BINDIR}/egrep \
 	${BINDIR}/grep ${BINDIR}/fgrep \
 	${BINDIR}/grep ${BINDIR}/zgrep \
 	${BINDIR}/grep ${BINDIR}/zegrep \
 	${BINDIR}/grep ${BINDIR}/zfgrep
 
 MLINKS= grep.1 egrep.1 \
 	grep.1 fgrep.1 \
 	grep.1 zgrep.1 \
 	grep.1 zegrep.1 \
-	grep.1 zfgrep.1 \
-	grep.1 xzgrep.1 \
-	grep.1 xzegrep.1 \
-	grep.1 xzfgrep.1 \
-	grep.1 lzgrep.1 \
-	grep.1 lzegrep.1 \
-	grep.1 lzfgrep.1
+	grep.1 zfgrep.1
 .endif
 
 LIBADD=	z
 
 .if ${MK_LZMA_SUPPORT} != "no"
 LIBADD+=	lzma
 
 LINKS+=	${BINDIR}/${PROG} ${BINDIR}/xzgrep \
 	${BINDIR}/${PROG} ${BINDIR}/xzegrep \
 	${BINDIR}/${PROG} ${BINDIR}/xzfgrep \
 	${BINDIR}/${PROG} ${BINDIR}/lzgrep \
 	${BINDIR}/${PROG} ${BINDIR}/lzegrep \
 	${BINDIR}/${PROG} ${BINDIR}/lzfgrep
+
+MLINKS+= grep.1 xzgrep.1 \
+	 grep.1 xzegrep.1 \
+	 grep.1 xzfgrep.1 \
+	 grep.1 lzgrep.1 \
+	 grep.1 lzegrep.1 \
+	 grep.1 lzfgrep.1
 .else
 CFLAGS+= -DWITHOUT_LZMA
 .endif
 
 .if ${MK_BZIP2_SUPPORT} != "no"
 LIBADD+=	bz2
 
 .if ${MK_BSD_GREP} == "yes"
 LINKS+= ${BINDIR}/grep ${BINDIR}/bzgrep \
 	${BINDIR}/grep ${BINDIR}/bzegrep \
 	${BINDIR}/grep ${BINDIR}/bzfgrep
 MLINKS+= grep.1 bzgrep.1 \
 	 grep.1 bzegrep.1 \
 	 grep.1 bzfgrep.1
 .endif
 .else
 CFLAGS+= -DWITHOUT_BZIP2
 .endif
 
 .if ${MK_GNU_GREP_COMPAT} != "no"
 CFLAGS+= -I${DESTDIR}/usr/include/gnu
 LIBADD+=	gnuregex
 .endif
 
 .if ${MK_NLS} != "no"
 .include "${.CURDIR}/nls/Makefile.inc"
 .else
 CFLAGS+= -DWITHOUT_NLS
 .endif
 
 .if ${MK_TESTS} != "no"
 SUBDIR+=	tests
 .endif
 
 .include <bsd.prog.mk>
Index: projects/clang360-import/usr.sbin/config/config.8
===================================================================
--- projects/clang360-import/usr.sbin/config/config.8	(revision 277944)
+++ projects/clang360-import/usr.sbin/config/config.8	(revision 277945)
@@ -1,270 +1,283 @@
 .\" Copyright (c) 1980, 1991, 1993
 .\"	The Regents of the University of California.  All rights reserved.
 .\"
 .\" Redistribution and use in source and binary forms, with or without
 .\" modification, are permitted provided that the following conditions
 .\" are met:
 .\" 1. Redistributions of source code must retain the above copyright
 .\"    notice, this list of conditions and the following disclaimer.
 .\" 2. Redistributions in binary form must reproduce the above copyright
 .\"    notice, this list of conditions and the following disclaimer in the
 .\"    documentation and/or other materials provided with the distribution.
 .\" 4. Neither the name of the University nor the names of its contributors
 .\"    may be used to endorse or promote products derived from this software
 .\"    without specific prior written permission.
 .\"
 .\" THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
 .\" ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
 .\" IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
 .\" ARE DISCLAIMED.  IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
 .\" FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
 .\" DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
 .\" OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
 .\" HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
 .\" LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
 .\" OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
 .\" SUCH DAMAGE.
 .\"
 .\"     @(#)config.8	8.2 (Berkeley) 4/19/94
 .\" $FreeBSD$
 .\"
 .Dd May 8, 2007
 .Dt CONFIG 8
 .Os
 .Sh NAME
 .Nm config
 .Nd build system configuration files
 .Sh SYNOPSIS
 .Nm
 .Op Fl CVgp
 .Op Fl I Ar path
 .Op Fl d Ar destdir
+.Op Fl s Ar srcdir
 .Ar SYSTEM_NAME
 .Nm
 .Op Fl x Ar kernel
 .Sh DESCRIPTION
 The
 .Nm
 utility builds a set of system configuration files from the file
 .Ar SYSTEM_NAME
 which describes
 the system to configure.
 A second file
 tells
 .Nm
 what files are needed to generate a system and
 can be augmented by configuration specific set of files
 that give alternate files for a specific machine
 (see the
 .Sx FILES
 section below).
 .Pp
 Available options and operands:
 .Bl -tag -width ".Ar SYSTEM_NAME"
 .It Fl V
 Print the
 .Nm
 version number.
 .It Fl C
 If the INCLUDE_CONFIG_FILE is present in a configuration file,
 kernel image will contain full configuration files included
 literally (preserving comments).
 This flag is kept for backward compatibility.
 .It Fl I Ar path
 Search in
 .Ar path
 for any file included by the
 .Ic include
 directive.  This option may be specified more than once.
 .It Fl d Ar destdir
 Use
 .Ar destdir
 as the output directory, instead of the default one.
 Note that
 .Nm
 does not append
 .Ar SYSTEM_NAME
 to the directory given.
+.It Fl s Ar srcdir
+Use
+.Ar srcdir
+as the source directory, instead of the default one.
 .It Fl m
 Print the MACHINE and MACHINE_ARCH values for this
 kernel and exit.
 .It Fl g
 Configure a system for debugging.
 .It Fl x Ar kernel
 Print kernel configuration file embedded into a kernel
 file.
 This option makes sense only if
 .Cd "options INCLUDE_CONFIG_FILE"
 entry was present in your configuration file.
 .It Fl p
 Configure a system for profiling; for example,
 .Xr kgmon 8
 and
 .Xr gprof 1 .
 If two or more
 .Fl p
 options are supplied,
 .Nm
 configures a system for high resolution profiling.
 .It Ar SYSTEM_NAME
 Specify the name of the system configuration file
 containing device specifications, configuration options
 and other system parameters for one system configuration.
 .El
 .Pp
 The
 .Nm
 utility should be run from the
 .Pa conf
 subdirectory of the system source (usually
 .Pa /sys/ Ns Va ARCH Ns Pa /conf ) ,
 where
 .Va ARCH
 represents one of the architectures supported by
 .Fx .
 The
 .Nm
 utility creates the directory
 .Pa ../compile/ Ns Ar SYSTEM_NAME
 or the one given with the
 .Fl d
 option
 as necessary and places all output files there.
 The output of
 .Nm
 consists of a number of files; for the
 .Tn i386 ,
 they are:
 .Pa Makefile ,
 used by
 .Xr make 1
 in building the system;
 header files,
 definitions of
 the number of various devices that will be compiled into the system.
+.Pp
+The
+.Nm
+utility looks for kernel sources in the directory
+.Pa ../..
+or the one given with the
+.Fl s
+option.
 .Pp
 After running
 .Nm ,
 it is necessary to run
 .Dq Li make depend
 in the directory where the new makefile
 was created.
 The
 .Nm
 utility prints a reminder of this when it completes.
 .Pp
 If any other error messages are produced by
 .Nm ,
 the problems in the configuration file should be corrected and
 .Nm
 should be run again.
 Attempts to compile a system that had configuration errors
 are likely to fail.
 .Sh DEBUG KERNELS
 Traditional
 .Bx
 kernels are compiled without symbols due to the heavy load on the
 system when compiling a
 .Dq debug
 kernel.
 A debug kernel contains complete symbols for all the source files, and
 enables an experienced kernel programmer to analyse the cause of a problem.
 The
 debuggers available prior to
 .Bx 4.4 Lite
 were able to find some information
 from a normal kernel;
 .Xr gdb 1
 provides very little support for normal kernels, and a debug kernel is needed
 for any meaningful analysis.
 .Pp
 For reasons of history, time and space, building a debug kernel is not the
 default with
 .Fx :
 a debug kernel takes up to 30% longer to build and
 requires about 30 MB of disk storage in the build directory, compared to about 6
 MB for a non-debug kernel.
 A debug kernel is about 11 MB in size, compared to
 about 2 MB for a non-debug kernel.
 This space is used both in the root file
 system and at run time in memory.
 Use the
 .Fl g
 option to build a debug kernel.
 With this option,
 .Nm
 causes two kernel files to be built in the kernel build directory:
 .Bl -bullet
 .It
 .Pa kernel.debug
 is the complete debug kernel.
 .It
 .Pa kernel
 is a copy of the kernel with the debug symbols stripped off.
 This is equivalent
 to the normal non-debug kernel.
 .El
 .Pp
 There is currently little sense in installing and booting from a debug kernel,
 since the only tools available which use the symbols do not run on-line.
 There
 are therefore two options for installing a debug kernel:
 .Bl -bullet
 .It
 .Dq Li "make install"
 installs
 .Pa kernel
 in the root file system.
 .It
 .Dq Li "make install.debug"
 installs
 .Pa kernel.debug
 in the root file system.
 .El
 .Sh FILES
 .Bl -tag -width ".Pa /sys/ Ns Va ARCH Ns Pa /compile/ Ns Ar SYSTEM_NAME" -compact
 .It Pa /sys/conf/files
 list of common files system is built from
 .It Pa /sys/conf/Makefile. Ns Va ARCH
 generic makefile for the
 .Va ARCH
 .It Pa /sys/conf/files. Ns Va ARCH
 list of
 .Va ARCH
 specific files
 .It Pa /sys/ Ns Va ARCH Ns Pa /compile/ Ns Ar SYSTEM_NAME
 default kernel build directory for system
 .Ar SYSTEM_NAME
 on
 .Va ARCH .
 .El
 .Sh SEE ALSO
 .Xr config 5
 .Pp
 The
 .Sx SYNOPSIS
 portion of each device in section 4.
 .Rs
 .%T "Building 4.3 BSD UNIX System with Config"
 .Re
 .Sh HISTORY
 The
 .Nm
 utility appeared in
 .Bx 4.1 .
 .Pp
 Before support for
 .Fl x
 was introduced,
 .Cd "options INCLUDE_CONFIG_FILE"
 included entire configuration file that used to be embedded in
 the new kernel.
 This meant that
 .Xr strings 1
 could be used to extract it from a kernel:
 to extract the configuration information, you had to use
 the command:
 .Pp
 .Dl "strings -n 3 kernel | sed -n 's/^___//p'"
 .Sh BUGS
 The line numbers reported in error messages are usually off by one.
Index: projects/clang360-import/usr.sbin/config/main.c
===================================================================
--- projects/clang360-import/usr.sbin/config/main.c	(revision 277944)
+++ projects/clang360-import/usr.sbin/config/main.c	(revision 277945)
@@ -1,770 +1,778 @@
 /*
  * Copyright (c) 1980, 1993
  *	The Regents of the University of California.  All rights reserved.
  *
  * Redistribution and use in source and binary forms, with or without
  * modification, are permitted provided that the following conditions
  * are met:
  * 1. Redistributions of source code must retain the above copyright
  *    notice, this list of conditions and the following disclaimer.
  * 2. Redistributions in binary form must reproduce the above copyright
  *    notice, this list of conditions and the following disclaimer in the
  *    documentation and/or other materials provided with the distribution.
  * 4. Neither the name of the University nor the names of its contributors
  *    may be used to endorse or promote products derived from this software
  *    without specific prior written permission.
  *
  * THIS SOFTWARE IS PROVIDED BY THE REGENTS AND CONTRIBUTORS ``AS IS'' AND
  * ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
  * IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
  * ARE DISCLAIMED.  IN NO EVENT SHALL THE REGENTS OR CONTRIBUTORS BE LIABLE
  * FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
  * DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
  * OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
  * HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT
  * LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY
  * OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF
  * SUCH DAMAGE.
  */
 
 #ifndef lint
 static const char copyright[] =
 "@(#) Copyright (c) 1980, 1993\n\
 	The Regents of the University of California.  All rights reserved.\n";
 #endif /* not lint */
 
 #ifndef lint
 #if 0
 static char sccsid[] = "@(#)main.c	8.1 (Berkeley) 6/6/93";
 #endif
 static const char rcsid[] =
   "$FreeBSD$";
 #endif /* not lint */
 
 #include <sys/types.h>
 #include <sys/stat.h>
 #include <sys/sbuf.h>
 #include <sys/file.h>
 #include <sys/mman.h>
 #include <sys/param.h>
 
 #include <assert.h>
 #include <ctype.h>
 #include <err.h>
 #include <stdio.h>
 #include <string.h>
 #include <sysexits.h>
 #include <unistd.h>
 #include <dirent.h>
 #include "y.tab.h"
 #include "config.h"
 #include "configvers.h"
 
 #ifndef TRUE
 #define TRUE	(1)
 #endif
 
 #ifndef FALSE
 #define FALSE	(0)
 #endif
 
 #define	CDIR	"../compile/"
 
 char *	PREFIX;
 char 	destdir[MAXPATHLEN];
 char 	srcdir[MAXPATHLEN];
 
 int	debugging;
 int	profiling;
 int	found_defaults;
 int	incignore;
 
 /*
  * Preserve old behaviour in INCLUDE_CONFIG_FILE handling (files are included
  * literally).
  */
 int	filebased = 0;
 
 static void configfile(void);
 static void get_srcdir(void);
 static void usage(void);
 static void cleanheaders(char *);
 static void kernconfdump(const char *);
 static void checkversion(void);
 extern int yyparse(void);
 
 struct hdr_list {
 	char *h_name;
 	struct hdr_list *h_next;
 } *htab;
 
 /*
  * Config builds a set of files for building a UNIX
  * system given a description of the desired system.
  */
 int
 main(int argc, char **argv)
 {
 
 	struct stat buf;
 	int ch, len;
 	char *p;
 	char *kernfile;
 	struct includepath* ipath;
 	int printmachine;
 
 	printmachine = 0;
 	kernfile = NULL;
 	SLIST_INIT(&includepath);
-	while ((ch = getopt(argc, argv, "CI:d:gmpVx:")) != -1)
+	while ((ch = getopt(argc, argv, "CI:d:gmpsVx:")) != -1)
 		switch (ch) {
 		case 'C':
 			filebased = 1;
 			break;
 		case 'I':
 			ipath = (struct includepath *) \
 			    	calloc(1, sizeof (struct includepath));
 			if (ipath == NULL)
 				err(EXIT_FAILURE, "calloc");
 			ipath->path = optarg;
 			SLIST_INSERT_HEAD(&includepath, ipath, path_next);
 			break;
 		case 'm':
 			printmachine = 1;
 			break;
 		case 'd':
 			if (*destdir == '\0')
 				strlcpy(destdir, optarg, sizeof(destdir));
 			else
 				errx(EXIT_FAILURE, "directory already set");
 			break;
 		case 'g':
 			debugging++;
 			break;
 		case 'p':
 			profiling++;
 			break;
+		case 's':
+			if (*srcdir == '\0')
+				strlcpy(srcdir, optarg, sizeof(srcdir));
+			else
+				errx(EXIT_FAILURE, "src directory already set");
+			break;
 		case 'V':
 			printf("%d\n", CONFIGVERS);
 			exit(0);
 		case 'x':
 			kernfile = optarg;
 			break;
 		case '?':
 		default:
 			usage();
 		}
 	argc -= optind;
 	argv += optind;
 
 	if (kernfile != NULL) {
 		kernconfdump(kernfile);
 		exit(EXIT_SUCCESS);
 	}
 
 	if (argc != 1)
 		usage();
 
 	PREFIX = *argv;
 	if (stat(PREFIX, &buf) != 0 || !S_ISREG(buf.st_mode))
 		err(2, "%s", PREFIX);
 	if (freopen("DEFAULTS", "r", stdin) != NULL) {
 		found_defaults = 1;
 		yyfile = "DEFAULTS";
 	} else {
 		if (freopen(PREFIX, "r", stdin) == NULL)
 			err(2, "%s", PREFIX);
 		yyfile = PREFIX;
 	}
 	if (*destdir != '\0') {
 		len = strlen(destdir);
 		while (len > 1 && destdir[len - 1] == '/')
 			destdir[--len] = '\0';
-		get_srcdir();
+		if (*srcdir == '\0')
+			get_srcdir();
 	} else {
 		strlcpy(destdir, CDIR, sizeof(destdir));
 		strlcat(destdir, PREFIX, sizeof(destdir));
 	}
 
 	SLIST_INIT(&cputype);
 	SLIST_INIT(&mkopt);
 	SLIST_INIT(&opt);
 	SLIST_INIT(&rmopts);
 	STAILQ_INIT(&cfgfiles);
 	STAILQ_INIT(&dtab);
 	STAILQ_INIT(&fntab);
 	STAILQ_INIT(&ftab);
 	STAILQ_INIT(&hints);
 	if (yyparse())
 		exit(3);
 
 	/*
 	 * Ensure that required elements (machine, cpu, ident) are present.
 	 */
 	if (machinename == NULL) {
 		printf("Specify machine type, e.g. ``machine i386''\n");
 		exit(1);
 	}
 	if (ident == NULL) {
 		printf("no ident line specified\n");
 		exit(1);
 	}
 	if (SLIST_EMPTY(&cputype)) {
 		printf("cpu type must be specified\n");
 		exit(1);
 	}
 	checkversion();
 
 	if (printmachine) {
 		printf("%s\t%s\n",machinename,machinearch);
 		exit(0);
 	}
 
 	/* Make compile directory */
 	p = path((char *)NULL);
 	if (stat(p, &buf)) {
 		if (mkdir(p, 0777))
 			err(2, "%s", p);
 	} else if (!S_ISDIR(buf.st_mode))
 		errx(EXIT_FAILURE, "%s isn't a directory", p);
 
 	configfile();			/* put config file into kernel*/
 	options();			/* make options .h files */
 	makefile();			/* build Makefile */
 	makeenv();			/* build env.c */
 	makehints();			/* build hints.c */
 	headers();			/* make a lot of .h files */
 	cleanheaders(p);
 	printf("Kernel build directory is %s\n", p);
 	printf("Don't forget to do ``make cleandepend && make depend''\n");
 	exit(0);
 }
 
 /*
  * get_srcdir
  *	determine the root of the kernel source tree
  *	and save that in srcdir.
  */
 static void
 get_srcdir(void)
 {
 	struct stat lg, phy;
 	char *p, *pwd;
 	int i;
 
 	if (realpath("../..", srcdir) == NULL)
 		err(EXIT_FAILURE, "Unable to find root of source tree");
 	if ((pwd = getenv("PWD")) != NULL && *pwd == '/' &&
 	    (pwd = strdup(pwd)) != NULL) {
 		/* Remove the last two path components. */
 		for (i = 0; i < 2; i++) {
 			if ((p = strrchr(pwd, '/')) == NULL) {
 				free(pwd);
 				return;
 			}
 			*p = '\0';
 		}
 		if (stat(pwd, &lg) != -1 && stat(srcdir, &phy) != -1 &&
 		    lg.st_dev == phy.st_dev && lg.st_ino == phy.st_ino)
 			strlcpy(srcdir, pwd, MAXPATHLEN);
 		free(pwd);
 	}
 }
 
 static void
 usage(void)
 {
 
-	fprintf(stderr, "usage: config [-CgmpV] [-d destdir] sysname\n");
+	fprintf(stderr,
+	    "usage: config [-CgmpV] [-d destdir] [-s srcdir] sysname\n");
 	fprintf(stderr, "       config -x kernel\n");
 	exit(EX_USAGE);
 }
 
 /*
  * get_word
  *	returns EOF on end of file
  *	NULL on end of line
  *	pointer to the word otherwise
  */
 char *
 get_word(FILE *fp)
 {
 	static char line[80];
 	int ch;
 	char *cp;
 	int escaped_nl = 0;
 
 begin:
 	while ((ch = getc(fp)) != EOF)
 		if (ch != ' ' && ch != '\t')
 			break;
 	if (ch == EOF)
 		return ((char *)EOF);
 	if (ch == '\\'){
 		escaped_nl = 1;
 		goto begin;
 	}
 	if (ch == '\n') {
 		if (escaped_nl){
 			escaped_nl = 0;
 			goto begin;
 		}
 		else
 			return (NULL);
 	}
 	cp = line;
 	*cp++ = ch;
 	/* Negation operator is a word by itself. */
 	if (ch == '!') {
 		*cp = 0;
 		return (line);
 	}
 	while ((ch = getc(fp)) != EOF) {
 		if (isspace(ch))
 			break;
 		*cp++ = ch;
 	}
 	*cp = 0;
 	if (ch == EOF)
 		return ((char *)EOF);
 	(void) ungetc(ch, fp);
 	return (line);
 }
 
 /*
  * get_quoted_word
  *	like get_word but will accept something in double or single quotes
  *	(to allow embedded spaces).
  */
 char *
 get_quoted_word(FILE *fp)
 {
 	static char line[256];
 	int ch;
 	char *cp;
 	int escaped_nl = 0;
 
 begin:
 	while ((ch = getc(fp)) != EOF)
 		if (ch != ' ' && ch != '\t')
 			break;
 	if (ch == EOF)
 		return ((char *)EOF);
 	if (ch == '\\'){
 		escaped_nl = 1;
 		goto begin;
 	}
 	if (ch == '\n') {
 		if (escaped_nl){
 			escaped_nl = 0;
 			goto begin;
 		}
 		else
 			return (NULL);
 	}
 	cp = line;
 	if (ch == '"' || ch == '\'') {
 		int quote = ch;
 
 		escaped_nl = 0;
 		while ((ch = getc(fp)) != EOF) {
 			if (ch == quote && !escaped_nl)
 				break;
 			if (ch == '\n' && !escaped_nl) {
 				*cp = 0;
 				printf("config: missing quote reading `%s'\n",
 					line);
 				exit(2);
 			}
 			if (ch == '\\' && !escaped_nl) {
 				escaped_nl = 1;
 				continue;
 			}
 			if (ch != quote && escaped_nl)
 				*cp++ = '\\';
 			*cp++ = ch;
 			escaped_nl = 0;
 		}
 	} else {
 		*cp++ = ch;
 		while ((ch = getc(fp)) != EOF) {
 			if (isspace(ch))
 				break;
 			*cp++ = ch;
 		}
 		if (ch != EOF)
 			(void) ungetc(ch, fp);
 	}
 	*cp = 0;
 	if (ch == EOF)
 		return ((char *)EOF);
 	return (line);
 }
 
 /*
  * prepend the path to a filename
  */
 char *
 path(const char *file)
 {
 	char *cp = NULL;
 
 	if (file)
 		asprintf(&cp, "%s/%s", destdir, file);
 	else
 		cp = strdup(destdir);
 	return (cp);
 }
 
 /*
  * Generate configuration file based on actual settings. With this mode, user
  * will be able to obtain and build conifguration file with one command.
  */
 static void
 configfile_dynamic(struct sbuf *sb)
 {
 	struct cputype *cput;
 	struct device *d;
 	struct opt *ol;
 	char *lend;
 	unsigned int i;
 
 	asprintf(&lend, "\\n\\\n");
 	assert(lend != NULL);
 	sbuf_printf(sb, "options\t%s%s", OPT_AUTOGEN, lend);
 	sbuf_printf(sb, "ident\t%s%s", ident, lend);
 	sbuf_printf(sb, "machine\t%s%s", machinename, lend);
 	SLIST_FOREACH(cput, &cputype, cpu_next)
 		sbuf_printf(sb, "cpu\t%s%s", cput->cpu_name, lend);
 	SLIST_FOREACH(ol, &mkopt, op_next)
 		sbuf_printf(sb, "makeoptions\t%s=%s%s", ol->op_name,
 		    ol->op_value, lend);
 	SLIST_FOREACH(ol, &opt, op_next) {
 		if (strncmp(ol->op_name, "DEV_", 4) == 0)
 			continue;
 		sbuf_printf(sb, "options\t%s", ol->op_name);
 		if (ol->op_value != NULL) {
 			sbuf_putc(sb, '=');
 			for (i = 0; i < strlen(ol->op_value); i++) {
 				if (ol->op_value[i] == '"')
 					sbuf_printf(sb, "\\%c",
 					    ol->op_value[i]);
 				else
 					sbuf_printf(sb, "%c",
 					    ol->op_value[i]);
 			}
 			sbuf_printf(sb, "%s", lend);
 		} else {
 			sbuf_printf(sb, "%s", lend);
 		}
 	}
 	/*
 	 * Mark this file as containing everything we need.
 	 */
 	STAILQ_FOREACH(d, &dtab, d_next)
 		sbuf_printf(sb, "device\t%s%s", d->d_name, lend);
 	free(lend);
 }
 
 /*
  * Generate file from the configuration files.
  */
 static void
 configfile_filebased(struct sbuf *sb)
 {
 	FILE *cff;
 	struct cfgfile *cf;
 	int i;
 
 	/*
 	 * Try to read all configuration files. Since those will be present as
 	 * C string in the macro, we have to slash their ends then the line
 	 * wraps.
 	 */
 	STAILQ_FOREACH(cf, &cfgfiles, cfg_next) {
 		cff = fopen(cf->cfg_path, "r");
 		if (cff == NULL) {
 			warn("Couldn't open file %s", cf->cfg_path);
 			continue;
 		}
 		while ((i = getc(cff)) != EOF) {
 			if (i == '\n')
 				sbuf_printf(sb, "\\n\\\n");
 			else if (i == '"' || i == '\'')
 				sbuf_printf(sb, "\\%c", i);
 			else
 				sbuf_putc(sb, i);
 		}
 		fclose(cff);
 	}
 }
 
 static void
 configfile(void)
 {
 	FILE *fo;
 	struct sbuf *sb;
 	char *p;
 
 	/* Add main configuration file to the list of files to be included */
 	cfgfile_add(PREFIX);
 	p = path("config.c.new");
 	fo = fopen(p, "w");
 	if (!fo)
 		err(2, "%s", p);
 	sb = sbuf_new(NULL, NULL, 2048, SBUF_AUTOEXTEND);
 	assert(sb != NULL);
 	sbuf_clear(sb);
 	if (filebased) {
 		/* Is needed, can be used for backward compatibility. */
 		configfile_filebased(sb);
 	} else {
 		configfile_dynamic(sb);
 	}
 	sbuf_finish(sb);
 	/* 
 	 * We print first part of the template, replace our tag with
 	 * configuration files content and later continue writing our
 	 * template.
 	 */
 	p = strstr(kernconfstr, KERNCONFTAG);
 	if (p == NULL)
 		errx(EXIT_FAILURE, "Something went terribly wrong!");
 	*p = '\0';
 	fprintf(fo, "%s", kernconfstr);
 	fprintf(fo, "%s", sbuf_data(sb));
 	p += strlen(KERNCONFTAG);
 	fprintf(fo, "%s", p);
 	sbuf_delete(sb);
 	fclose(fo);
 	moveifchanged(path("config.c.new"), path("config.c"));
 	cfgfile_removeall();
 }
 
 /*
  * moveifchanged --
  *	compare two files; rename if changed.
  */
 void
 moveifchanged(const char *from_name, const char *to_name)
 {
 	char *p, *q;
 	int changed;
 	size_t tsize;
 	struct stat from_sb, to_sb;
 	int from_fd, to_fd;
 
 	changed = 0;
 
 	if ((from_fd = open(from_name, O_RDONLY)) < 0)
 		err(EX_OSERR, "moveifchanged open(%s)", from_name);
 
 	if ((to_fd = open(to_name, O_RDONLY)) < 0)
 		changed++;
 
 	if (!changed && fstat(from_fd, &from_sb) < 0)
 		err(EX_OSERR, "moveifchanged fstat(%s)", from_name);
 
 	if (!changed && fstat(to_fd, &to_sb) < 0)
 		err(EX_OSERR, "moveifchanged fstat(%s)", to_name);
 
 	if (!changed && from_sb.st_size != to_sb.st_size)
 		changed++;
 
 	tsize = (size_t)from_sb.st_size;
 
 	if (!changed) {
 		p = mmap(NULL, tsize, PROT_READ, MAP_SHARED, from_fd, (off_t)0);
 		if (p == MAP_FAILED)
 			err(EX_OSERR, "mmap %s", from_name);
 		q = mmap(NULL, tsize, PROT_READ, MAP_SHARED, to_fd, (off_t)0);
 		if (q == MAP_FAILED)
 			err(EX_OSERR, "mmap %s", to_name);
 
 		changed = memcmp(p, q, tsize);
 		munmap(p, tsize);
 		munmap(q, tsize);
 	}
 	if (changed) {
 		if (rename(from_name, to_name) < 0)
 			err(EX_OSERR, "rename(%s, %s)", from_name, to_name);
 	} else {
 		if (unlink(from_name) < 0)
 			err(EX_OSERR, "unlink(%s)", from_name);
 	}
 }
 
 static void
 cleanheaders(char *p)
 {
 	DIR *dirp;
 	struct dirent *dp;
 	struct file_list *fl;
 	struct hdr_list *hl;
 	size_t len;
 
 	remember("y.tab.h");
 	remember("setdefs.h");
 	STAILQ_FOREACH(fl, &ftab, f_next)
 		remember(fl->f_fn);
 
 	/*
 	 * Scan the build directory and clean out stuff that looks like
 	 * it might have been a leftover NFOO header, etc.
 	 */
 	if ((dirp = opendir(p)) == NULL)
 		err(EX_OSERR, "opendir %s", p);
 	while ((dp = readdir(dirp)) != NULL) {
 		len = strlen(dp->d_name);
 		/* Skip non-headers */
 		if (len < 2 || dp->d_name[len - 2] != '.' ||
 		    dp->d_name[len - 1] != 'h')
 			continue;
 		/* Skip special stuff, eg: bus_if.h, but check opt_*.h */
 		if (strchr(dp->d_name, '_') &&
 		    strncmp(dp->d_name, "opt_", 4) != 0)
 			continue;
 		/* Check if it is a target file */
 		for (hl = htab; hl != NULL; hl = hl->h_next) {
 			if (eq(dp->d_name, hl->h_name)) {
 				break;
 			}
 		}
 		if (hl)
 			continue;
 		printf("Removing stale header: %s\n", dp->d_name);
 		if (unlink(path(dp->d_name)) == -1)
 			warn("unlink %s", dp->d_name);
 	}
 	(void)closedir(dirp);
 }
 
 void
 remember(const char *file)
 {
 	char *s;
 	struct hdr_list *hl;
 
 	if ((s = strrchr(file, '/')) != NULL)
 		s = ns(s + 1);
 	else
 		s = ns(file);
 
 	if (strchr(s, '_') && strncmp(s, "opt_", 4) != 0) {
 		free(s);
 		return;
 	}
 	for (hl = htab; hl != NULL; hl = hl->h_next) {
 		if (eq(s, hl->h_name)) {
 			free(s);
 			return;
 		}
 	}
 	hl = calloc(1, sizeof(*hl));
 	if (hl == NULL)
 		err(EXIT_FAILURE, "calloc");
 	hl->h_name = s;
 	hl->h_next = htab;
 	htab = hl;
 }
 
 /*
  * This one is quick hack. Will be probably moved to elf(3) interface.
  * It takes kernel configuration file name, passes it as an argument to
  * elfdump -a, which output is parsed by some UNIX tools...
  */
 static void
 kernconfdump(const char *file)
 {
 	struct stat st;
 	FILE *fp, *pp;
 	int error, len, osz, r;
 	unsigned int i, off, size, t1, t2, align;
 	char *cmd, *o;
 
 	r = open(file, O_RDONLY);
 	if (r == -1)
 		err(EXIT_FAILURE, "Couldn't open file '%s'", file);
 	error = fstat(r, &st);
 	if (error == -1)
 		err(EXIT_FAILURE, "fstat() failed");
 	if (S_ISDIR(st.st_mode))
 		errx(EXIT_FAILURE, "'%s' is a directory", file);
 	fp = fdopen(r, "r");
 	if (fp == NULL)
 		err(EXIT_FAILURE, "fdopen() failed");
 	osz = 1024;
 	o = calloc(1, osz);
 	if (o == NULL)
 		err(EXIT_FAILURE, "Couldn't allocate memory");
 	/* ELF note section header. */
 	asprintf(&cmd, "/usr/bin/elfdump -c %s | grep -A 8 kern_conf"
 	    "| tail -5 | cut -d ' ' -f 2 | paste - - - - -", file);
 	if (cmd == NULL)
 		errx(EXIT_FAILURE, "asprintf() failed");
 	pp = popen(cmd, "r");
 	if (pp == NULL)
 		errx(EXIT_FAILURE, "popen() failed");
 	free(cmd);
 	len = fread(o, osz, 1, pp);
 	pclose(pp);
 	r = sscanf(o, "%d%d%d%d%d", &off, &size, &t1, &t2, &align);
 	free(o);
 	if (r != 5)
 		errx(EXIT_FAILURE, "File %s doesn't contain configuration "
 		    "file. Either unsupported, or not compiled with "
 		    "INCLUDE_CONFIG_FILE", file);
 	r = fseek(fp, off, SEEK_CUR);
 	if (r != 0)
 		err(EXIT_FAILURE, "fseek() failed");
 	for (i = 0; i < size; i++) {
 		r = fgetc(fp);
 		if (r == EOF)
 			break;
 		if (r == '\0') {
 			assert(i == size - 1 &&
 			    ("\\0 found in the middle of a file"));
 			break;
 		}
 		fputc(r, stdout);
 	}
 	fclose(fp);
 }
 
 static void 
 badversion(int versreq)
 {
 	fprintf(stderr, "ERROR: version of config(8) does not match kernel!\n");
 	fprintf(stderr, "config version = %d, ", CONFIGVERS);
 	fprintf(stderr, "version required = %d\n\n", versreq);
 	fprintf(stderr, "Make sure that /usr/src/usr.sbin/config is in sync\n");
 	fprintf(stderr, "with your /usr/src/sys and install a new config binary\n");
 	fprintf(stderr, "before trying this again.\n\n");
 	fprintf(stderr, "If running the new config fails check your config\n");
 	fprintf(stderr, "file against the GENERIC or LINT config files for\n");
 	fprintf(stderr, "changes in config syntax, or option/device naming\n");
 	fprintf(stderr, "conventions\n\n");
 	exit(1);
 }
 
 static void
 checkversion(void)
 {
 	FILE *ifp;
 	char line[BUFSIZ];
 	int versreq;
 
 	ifp = open_makefile_template();
 	while (fgets(line, BUFSIZ, ifp) != 0) {
 		if (*line != '%')
 			continue;
 		if (strncmp(line, "%VERSREQ=", 9) != 0)
 			continue;
 		versreq = atoi(line + 9);
 		if (MAJOR_VERS(versreq) == MAJOR_VERS(CONFIGVERS) &&
 		    versreq <= CONFIGVERS)
 			continue;
 		badversion(versreq);
 	}
 	fclose(ifp);
 }
Index: projects/clang360-import
===================================================================
--- projects/clang360-import	(revision 277944)
+++ projects/clang360-import	(revision 277945)

Property changes on: projects/clang360-import
___________________________________________________________________
Modified: svn:mergeinfo
## -0,0 +0,1 ##
   Merged /head:r277902-277944