git: fc6ed8627222 - main - zfs: cherry-pick from openzfs/master (restore reverted commit)
- Go to: [ bottom of page ] [ top of archives ] [ this month ]
Date: Thu, 01 Oct 2026 09:54:37 UTC
The branch main has been updated by mm:
URL: https://cgit.FreeBSD.org/src/commit/?id=fc6ed8627222a625a700e99cdfcda19654a0c651
commit fc6ed8627222a625a700e99cdfcda19654a0c651
Author: Tony Hutter <hutter2@llnl.gov>
AuthorDate: 2026-09-16 22:12:23 +0000
Commit: Martin Matuska <mm@FreeBSD.org>
CommitDate: 2026-10-01 09:53:35 +0000
zfs: cherry-pick from openzfs/master (restore reverted commit)
zpool: Add zpool status -vv error ranges (#17864)
Add necessery change to libzfs/Makefile
(cherry picked from commit e903655c50fe2e710a4cdfe6e638454012ef6324)
---
cddl/lib/libzfs/Makefile | 6 +
sys/contrib/openzfs/cmd/zinject/translate.c | 79 ++++-
sys/contrib/openzfs/cmd/zpool/zpool_main.c | 239 ++++++++++++----
sys/contrib/openzfs/include/sys/fs/zfs.h | 18 ++
sys/contrib/openzfs/include/sys/spa.h | 23 ++
sys/contrib/openzfs/include/sys/zfs_stat.h | 23 +-
sys/contrib/openzfs/lib/libzfs/libzfs_pool.c | 317 +++++++++++++++++++--
sys/contrib/openzfs/lib/libzfs/libzfs_util.c | 2 +
sys/contrib/openzfs/man/man8/zinject.8 | 5 +-
sys/contrib/openzfs/man/man8/zpool-status.8 | 13 +-
.../openzfs/module/os/freebsd/zfs/zvol_os.c | 10 +
sys/contrib/openzfs/module/zfs/zfs_ioctl.c | 24 +-
sys/contrib/openzfs/module/zfs/zfs_znode.c | 49 +++-
sys/contrib/openzfs/module/zfs/zio.c | 2 -
sys/contrib/openzfs/tests/runfiles/common.run | 3 +-
.../openzfs/tests/zfs-tests/include/tunables.cfg | 1 +
.../openzfs/tests/zfs-tests/tests/Makefile.am | 1 +
.../cli_root/zpool_status/zpool_status_-v.ksh | 173 +++++++++++
18 files changed, 891 insertions(+), 97 deletions(-)
diff --git a/cddl/lib/libzfs/Makefile b/cddl/lib/libzfs/Makefile
index 83f93ec4276b..1749123c5cb8 100644
--- a/cddl/lib/libzfs/Makefile
+++ b/cddl/lib/libzfs/Makefile
@@ -1,5 +1,6 @@
.PATH: ${ZFSTOP}/module/icp
.PATH: ${ZFSTOP}/module/zcommon
+.PATH: ${ZFSTOP}/module/zfs
.PATH: ${ZFSTOP}/lib/libzfs
.PATH: ${ZFSTOP}/lib/libzfs/os/freebsd
.PATH: ${ZFSTOP}/include
@@ -52,7 +53,9 @@ USER_C += \
libzfs_zmount.c
KERNEL_C = \
+ btree.c \
cityhash.c \
+ range_tree.c \
zfeature_common.c \
zfs_comutil.c \
zfs_deleg.c \
@@ -105,4 +108,7 @@ CFLAGS+= -DSYSCONFDIR=\"/etc\"
CFLAGS+= -DPKGDATADIR=\"/usr/share/zfs\"
CFLAGS+= -DZFSEXECDIR=\"${LIBEXECDIR}/zfs\"
+CFLAGS.btree.c+=-fvisibility=hidden
+CFLAGS.range_tree.c+= -fvisibility=hidden
+
.include <bsd.lib.mk>
diff --git a/sys/contrib/openzfs/cmd/zinject/translate.c b/sys/contrib/openzfs/cmd/zinject/translate.c
index 1d95f0f399c0..0bc40c251403 100644
--- a/sys/contrib/openzfs/cmd/zinject/translate.c
+++ b/sys/contrib/openzfs/cmd/zinject/translate.c
@@ -15,6 +15,7 @@
*/
#include <libzfs.h>
+#include <libzutil.h>
#include <errno.h>
#include <fcntl.h>
@@ -33,6 +34,7 @@
#include <sys/dmu_objset.h>
#include <sys/dnode.h>
#include <sys/vdev_impl.h>
+#include <sys/zvol.h>
#include <sys/mkdev.h>
@@ -66,10 +68,46 @@ compress_slashes(const char *src, char *dest)
}
/*
- * Given a full path to a file, translate into a dataset name and a relative
- * path within the dataset. 'dataset' must be at least MAXNAMELEN characters,
- * and 'relpath' must be at least MAXPATHLEN characters. We also pass a stat64
- * buffer, which we need later to get the object ID.
+ * If 'inpath' points to a zvol block device, copy the zvol dataset path to 'ds'
+ * (like "tank/vol"), and return true. If 'inpath' does not point to a zvol
+ * block device, then return false.
+ */
+static boolean_t
+get_zvol_dataset(const char *inpath, char *ds)
+{
+ boolean_t rc = B_FALSE;
+
+ char *buf = realpath(inpath, NULL);
+ if (buf == NULL)
+ return (B_FALSE);
+
+ int fd = open(buf, O_RDONLY | O_CLOEXEC);
+ free(buf);
+
+ if (fd == -1)
+ return (B_FALSE);
+
+ /*
+ * We want to determine if 'inpath' is a zvol device. There's no
+ * universal way to do that, so we use OS-specific zvol ioctls.
+ */
+#if defined(__FreeBSD__)
+ if (ioctl(fd, DIOCGPHYSPATH, ds) == 0)
+ rc = B_TRUE;
+#else
+ if (ioctl(fd, BLKZNAME, ds) == 0)
+ rc = B_TRUE;
+#endif
+ close(fd);
+
+ return (rc);
+}
+
+/*
+ * Given a full path to a file or zvol device, translate into a dataset name and
+ * a relative path within the dataset. 'dataset' must be at least MAXNAMELEN
+ * characters, and 'relpath' must be at least MAXPATHLEN characters. We also
+ * pass a stat64 buffer, which we need later to get the object ID.
*/
static int
parse_pathname(const char *inpath, char *dataset, char *relpath,
@@ -88,6 +126,39 @@ parse_pathname(const char *inpath, char *dataset, char *relpath,
return (-1);
}
+ /* special case: inject errors into zvol */
+ if (get_zvol_dataset(inpath, fullpath)) {
+ char *slash;
+ /*
+ * The zvol's inode does not contain its object number.
+ * However, it has long been the case that the zvol data is
+ * object number 1 (ZVOL_OBJ):
+ *
+ * Object lvl iblk dblk dsize lsize %full type
+ * 0 6 128K 16K 11K 16K 6.25 DMU dnode
+ * 1 2 128K 16K 20.1M 20M 100.00 zvol object
+ * 2 1 128K 512 0 512 100.00 zvol prop
+ *
+ * So we hardcode that in the statbuf inode field as workaround.
+ */
+ statbuf->st_ino = ZVOL_OBJ;
+
+ (void) strlcpy(dataset, fullpath, MAXNAMELEN);
+
+ /*
+ * fullpath contains string like 'tank/zvol'. Strip off the
+ * 'tank' and 'zvol' parts.
+ */
+ slash = strchr(fullpath, '/');
+ if (slash == NULL) {
+ (void) fprintf(stderr, "invalid volume name: '%s'\n",
+ fullpath);
+ return (-1);
+ }
+ (void) strlcpy(relpath, slash + 1, MAXPATHLEN);
+ return (0);
+ }
+
if (getextmntent(fullpath, &mp, statbuf) != 0) {
(void) fprintf(stderr, "cannot find mountpoint for '%s'\n",
fullpath);
diff --git a/sys/contrib/openzfs/cmd/zpool/zpool_main.c b/sys/contrib/openzfs/cmd/zpool/zpool_main.c
index 10c0f8c7635b..40e954e1c033 100644
--- a/sys/contrib/openzfs/cmd/zpool/zpool_main.c
+++ b/sys/contrib/openzfs/cmd/zpool/zpool_main.c
@@ -527,7 +527,7 @@ get_usage(zpool_help_t idx)
return (gettext("\ttrim [-dw] [-r <rate>] [-c | -s] "
"<-a | <pool> [<device> ...]>\n"));
case HELP_STATUS:
- return (gettext("\tstatus [-DdegiLPpstvx] "
+ return (gettext("\tstatus [-DdegiLPpstx] [-v|-vv] "
"[-c script1[,script2,...]] ...\n"
"\t [-j|--json [--json-flat-vdevs] [--json-int] "
"[--json-pool-key-guid]] ...\n"
@@ -1075,6 +1075,9 @@ nice_num_str_nvlist(nvlist_t *item, const char *key, uint64_t value,
case ZFS_NICENUM_BYTES:
zfs_nicenum_format(value, buf, 256, ZFS_NICENUM_BYTES);
break;
+ case ZFS_NICENUM_RAW:
+ zfs_nicenum_format(value, buf, 256, ZFS_NICENUM_RAW);
+ break;
case ZFS_NICENUM_TIME:
zfs_nicenum_format(value, buf, 256, ZFS_NICENUM_TIME);
break;
@@ -2616,8 +2619,8 @@ typedef struct status_cbdata {
int cb_count;
int cb_name_flags;
int cb_namewidth;
+ int cb_verbosity;
boolean_t cb_allpools;
- boolean_t cb_verbose;
boolean_t cb_literal;
boolean_t cb_explain;
boolean_t cb_first;
@@ -3349,7 +3352,7 @@ print_class_vdevs(zpool_handle_t *zhp, status_cbdata_t *cb, nvlist_t *nv,
nvlist_t **child;
boolean_t printed = B_FALSE;
- assert(zhp != NULL || !cb->cb_verbose);
+ assert(zhp != NULL || cb->cb_verbosity == 0);
if (nvlist_lookup_nvlist_array(nv, ZPOOL_CONFIG_CHILDREN, &child,
&children) != 0)
@@ -9794,7 +9797,7 @@ class_vdevs_nvlist(zpool_handle_t *zhp, status_cbdata_t *cb, nvlist_t *nv,
if (!cb->cb_flat_vdevs)
class_obj = fnvlist_alloc();
- assert(zhp != NULL || !cb->cb_verbose);
+ assert(zhp != NULL || cb->cb_verbosity == 0);
if (nvlist_lookup_nvlist_array(nv, ZPOOL_CONFIG_CHILDREN, &child,
&children) != 0)
@@ -9898,57 +9901,133 @@ spares_nvlist(zpool_handle_t *zhp, status_cbdata_t *cb, nvlist_t *nv,
}
}
+/*
+ * Take a uint64 nvpair named 'name' from nverrlist, nicenum-ify it, and
+ * put it back in 'nverrlist', possibly as a string, with the same 'name'.
+ */
+static void
+convert_nvlist_uint64_to_nicenum(status_cbdata_t *cb, nvlist_t *parent,
+ const char *name, enum zfs_nicenum_format format)
+{
+ uint64_t val;
+ nvpair_t *nvp;
+
+ if (nvlist_lookup_nvpair(parent, name, &nvp) != 0)
+ return; /* nothing by that name, ignore */
+
+ val = fnvpair_value_uint64(nvp);
+ nvlist_remove_nvpair(parent, nvp);
+ nice_num_str_nvlist(parent, name, val,
+ cb->cb_literal, cb->cb_json_as_int, format);
+}
+
static void
errors_nvlist(zpool_handle_t *zhp, status_cbdata_t *cb, nvlist_t *item)
{
+ int verbosity = cb->cb_verbosity;
+ nvlist_t *nverrlist = NULL;
+ nvpair_t *elem;
+ char *pathname;
+ size_t len = MAXPATHLEN * 2;
+ nvlist_t **ranges;
+ uint_t count;
+ uint_t objs = 0, i;
+ nvlist_t **nv_arr = NULL;
+ char **str_arr = NULL;
uint64_t nerr;
- nvlist_t *config = zpool_get_config(zhp, NULL);
- if (nvlist_lookup_uint64(config, ZPOOL_CONFIG_ERRCOUNT,
- &nerr) == 0) {
- nice_num_str_nvlist(item, ZPOOL_CONFIG_ERRCOUNT, nerr,
+
+ if (zpool_get_errlog(zhp, &nverrlist) != 0) {
+ /* We're expected to always return an error count */
+ nice_num_str_nvlist(item, ZPOOL_CONFIG_ERRCOUNT, 0,
cb->cb_literal, cb->cb_json_as_int, ZFS_NICENUM_1024);
- if (nerr != 0 && cb->cb_verbose) {
- nvlist_t *nverrlist = NULL;
- if (zpool_get_errlog(zhp, &nverrlist) == 0) {
- int i = 0;
- int count = 0;
- size_t len = MAXPATHLEN * 2;
- nvpair_t *elem = NULL;
-
- for (nvpair_t *pair =
- nvlist_next_nvpair(nverrlist, NULL);
- pair != NULL;
- pair = nvlist_next_nvpair(nverrlist, pair))
- count++;
- char **errl = (char **)malloc(
- count * sizeof (char *));
-
- while ((elem = nvlist_next_nvpair(nverrlist,
- elem)) != NULL) {
- nvlist_t *nv;
- uint64_t dsobj, obj;
-
- verify(nvpair_value_nvlist(elem,
- &nv) == 0);
- verify(nvlist_lookup_uint64(nv,
- ZPOOL_ERR_DATASET, &dsobj) == 0);
- verify(nvlist_lookup_uint64(nv,
- ZPOOL_ERR_OBJECT, &obj) == 0);
- errl[i] = safe_malloc(len);
- zpool_obj_to_path(zhp, dsobj, obj,
- errl[i++], len);
- }
- nvlist_free(nverrlist);
- fnvlist_add_string_array(item, "errlist",
- (const char **)errl, count);
- for (int i = 0; i < count; ++i)
- free(errl[i]);
- free(errl);
- } else
- fnvlist_add_string(item, "errlist",
- strerror(errno));
+ return;
+ }
+
+ /*
+ * This 'error_count' entry is just the full error log block count.
+ * This includes duplicate and overlapping entries so it can be
+ * inaccurate. It's only included for historical reasons.
+ */
+ nvlist_t *config = zpool_get_config(zhp, NULL);
+ nerr = fnvlist_lookup_uint64(config, ZPOOL_CONFIG_ERRCOUNT);
+ nice_num_str_nvlist(item, ZPOOL_CONFIG_ERRCOUNT, nerr,
+ cb->cb_literal, cb->cb_json_as_int, ZFS_NICENUM_1024);
+
+ pathname = safe_malloc(len);
+
+ /* Get an initial count of all the objects (files) with errors */
+ elem = NULL;
+ while ((elem = nvlist_next_nvpair(nverrlist, elem)) != NULL)
+ objs++;
+
+ if (verbosity <= 1)
+ str_arr = safe_malloc(objs * sizeof (*str_arr));
+ else
+ nv_arr = safe_malloc(objs * sizeof (*nv_arr));
+
+ elem = NULL;
+ for (i = 0; i < objs; i++) {
+ nvlist_t *nv;
+ uint64_t dsobj, obj;
+ elem = nvlist_next_nvpair(nverrlist, elem);
+
+ nv = fnvpair_value_nvlist(elem);
+
+ dsobj = fnvlist_lookup_uint64(nv, ZPOOL_ERR_DATASET);
+ obj = fnvlist_lookup_uint64(nv, ZPOOL_ERR_OBJECT);
+
+ zpool_obj_to_path(zhp, dsobj, obj, pathname, len);
+
+ /*
+ * Each JSON entry is a different file/zvol. If user has
+ * verbosity = 1, then just we're just constructing an array
+ * of strings.
+ */
+ if (str_arr != NULL) {
+ str_arr[i] = strdup(pathname);
+ continue;
}
+
+ fnvlist_add_string(nv, ZPOOL_ERR_NAME, pathname);
+
+ /* nicenum-ify our nvlist */
+ convert_nvlist_uint64_to_nicenum(cb, nv, ZPOOL_ERR_OBJECT,
+ ZFS_NICENUM_RAW);
+ convert_nvlist_uint64_to_nicenum(cb, nv, ZPOOL_ERR_DATASET,
+ ZFS_NICENUM_RAW);
+ convert_nvlist_uint64_to_nicenum(cb, nv, ZPOOL_ERR_BLOCK_SIZE,
+ ZFS_NICENUM_1024);
+
+ if (nvlist_lookup_nvlist_array(nv, ZPOOL_ERR_RANGES, &ranges,
+ &count) == 0) {
+ for (uint_t i = 0; i < count; i++) {
+ convert_nvlist_uint64_to_nicenum(cb, ranges[i],
+ ZPOOL_ERR_START_BYTE, ZFS_NICENUM_1024);
+ convert_nvlist_uint64_to_nicenum(cb, ranges[i],
+ ZPOOL_ERR_END_BYTE, ZFS_NICENUM_1024);
+ }
+ }
+ nv_arr[i] = fnvlist_alloc();
+ fnvlist_add_nvlist(nv_arr[i], pathname, nv);
+ }
+
+ /* Place our error list in a top level "errlist" JSON array object. */
+ if (str_arr != NULL) {
+ fnvlist_add_string_array(item, ZPOOL_ERR_JSON,
+ (const char **) str_arr, objs);
+ for (i = 0; i < objs; i++)
+ free(str_arr[i]);
+ free(str_arr);
+ } else {
+ fnvlist_add_nvlist_array(item, ZPOOL_ERR_JSON,
+ (const nvlist_t **) nv_arr, objs);
+ for (i = 0; i < objs; i++)
+ nvlist_free(nv_arr[i]);
+ free(nv_arr);
}
+
+ free(pathname);
+ nvlist_free(nverrlist);
}
static void
@@ -10714,12 +10793,14 @@ print_condense_status(nvlist_t *nv)
}
static void
-print_error_log(zpool_handle_t *zhp)
+print_error_log(zpool_handle_t *zhp, int verbosity, boolean_t literal)
{
nvlist_t *nverrlist = NULL;
nvpair_t *elem;
char *pathname;
size_t len = MAXPATHLEN * 2;
+ boolean_t started = B_FALSE;
+ char last_pathname[MAXPATHLEN] = "";
if (zpool_get_errlog(zhp, &nverrlist) != 0)
return;
@@ -10747,8 +10828,51 @@ print_error_log(zpool_handle_t *zhp)
verify(nvlist_lookup_uint64(nv, ZPOOL_ERR_OBJECT,
&obj) == 0);
zpool_obj_to_path(zhp, dsobj, obj, pathname, len);
- (void) printf("%7s %s\n", "", pathname);
+ if (last_pathname[0] == '\0' ||
+ strncmp(pathname, last_pathname, sizeof (last_pathname))
+ != 0) {
+ strlcpy(last_pathname, pathname,
+ sizeof (last_pathname));
+ if (started)
+ (void) printf("\n");
+ else
+ started = B_TRUE;
+ (void) printf("%7s %s ", "", pathname);
+ } else if (verbosity > 1) {
+ (void) printf(",");
+ }
+ if (verbosity > 1) {
+ nvlist_t **arr;
+ uint_t count;
+ if (nvlist_lookup_nvlist_array(nv, ZPOOL_ERR_RANGES,
+ &arr, &count) != 0) {
+ (void) printf("(no ranges)");
+ continue;
+ }
+
+ for (uint_t i = 0; i < count; i++) {
+ uint64_t start;
+ uint64_t end;
+ start = fnvlist_lookup_uint64(arr[i],
+ ZPOOL_ERR_START_BYTE);
+ end = fnvlist_lookup_uint64(arr[i],
+ ZPOOL_ERR_END_BYTE);
+ if (literal) {
+ (void) printf("%llu-%llu",
+ (u_longlong_t)start,
+ (u_longlong_t)end);
+ } else {
+ char s1[32], s2[32];
+ zfs_nicenum(start, s1, sizeof (s1));
+ zfs_nicenum(end, s2, sizeof (s2));
+ (void) printf("%s-%s", s1, s2);
+ }
+ if (i != count - 1)
+ (void) printf(",");
+ }
+ }
}
+ (void) printf("\n");
free(pathname);
nvlist_free(nverrlist);
}
@@ -11330,7 +11454,13 @@ status_callback_json(zpool_handle_t *zhp, void *data)
spares_nvlist(zhp, cbp, nvroot, item);
}
dedup_stats_nvlist(zhp, cbp, item);
- errors_nvlist(zhp, cbp, item);
+
+ /*
+ * Historically, -j would always print the number of errors
+ * so check for that in addition to verbosity.
+ */
+ if (cbp->cb_verbosity > 0 || cbp->cb_json)
+ errors_nvlist(zhp, cbp, item);
}
if (cbp->cb_json_pool_key_guid) {
fnvlist_add_nvlist(d, pool_guid, item);
@@ -11504,14 +11634,15 @@ status_callback(zpool_handle_t *zhp, void *data)
if (nerr == 0) {
(void) printf(gettext(
"errors: No known data errors\n"));
- } else if (!cbp->cb_verbose) {
+ } else if (cbp->cb_verbosity == 0) {
color_start(ANSI_RED);
(void) printf(gettext("errors: %llu data "
"errors, use '-v' for details\n"),
(u_longlong_t)nerr);
color_end();
} else {
- print_error_log(zhp);
+ print_error_log(zhp, cbp->cb_verbosity,
+ cbp->cb_literal);
}
}
@@ -11638,7 +11769,7 @@ zpool_do_status(int argc, char **argv)
get_timestamp_arg(*optarg);
break;
case 'v':
- cb.cb_verbose = B_TRUE;
+ cb.cb_verbosity++;
break;
case 'j':
cb.cb_json = B_TRUE;
diff --git a/sys/contrib/openzfs/include/sys/fs/zfs.h b/sys/contrib/openzfs/include/sys/fs/zfs.h
index 11b1f67f33b7..da9801d4fa1b 100644
--- a/sys/contrib/openzfs/include/sys/fs/zfs.h
+++ b/sys/contrib/openzfs/include/sys/fs/zfs.h
@@ -1875,6 +1875,24 @@ typedef enum {
#define ZPOOL_ERR_LIST "error list"
#define ZPOOL_ERR_DATASET "dataset"
#define ZPOOL_ERR_OBJECT "object"
+#define ZPOOL_ERR_LEVEL "level"
+#define ZPOOL_ERR_BLKID "blkid" /* unused */
+
+/* Additional nvpairs from zpool_get_errlog() nvlist */
+#define ZPOOL_ERR_BLOCK_SIZE "block_size"
+#define ZPOOL_ERR_OBJECT_TYPE "object_type"
+#define ZPOOL_ERR_RANGES "ranges"
+#define ZPOOL_ERR_START_BYTE "start_byte"
+#define ZPOOL_ERR_END_BYTE "end_byte"
+#define ZPOOL_ERR_NAME "name"
+
+/*
+ * For the zpool status JSON output, we collect all the error lists and put
+ * them in a separate top level element so they're easier to iterate over.
+ * That way the error lists don't get interspersed with the zpool status
+ * objects.
+ */
+#define ZPOOL_ERR_JSON "errlist"
#define HIS_MAX_RECORD_LEN (MAXPATHLEN + MAXPATHLEN + 1)
diff --git a/sys/contrib/openzfs/include/sys/spa.h b/sys/contrib/openzfs/include/sys/spa.h
index 18fcbc831c8e..271768f9e6e0 100644
--- a/sys/contrib/openzfs/include/sys/spa.h
+++ b/sys/contrib/openzfs/include/sys/spa.h
@@ -354,6 +354,29 @@ typedef enum bp_embedded_type {
#define SPA_DVAS_PER_BP 3 /* Number of DVAs in a bp */
#define SPA_SYNC_MIN_VDEVS 3 /* min vdevs to update during sync */
+/*
+ * Get number of data block pointers an indirect block could point to (given
+ * the block level and block size shift).
+ *
+ * For example, an L1 block with a blocksize of 128kb could point to:
+ *
+ * BP_SPANB(17, 1) = 1024 L0 block pointers
+ */
+#define BP_SPANB(indblkshift, level) \
+ (((uint64_t)1) << ((level) * ((indblkshift) - SPA_BLKPTRSHIFT)))
+
+/*
+ * Helper function to lookup the byte range covered by a block pointer of any
+ * level (L0, L1, L2 ... etc).
+ *
+ * For example if you have an L1 block, and your blocksize is 128kb (shift 17),
+ * then your block can cover this many bytes:
+ *
+ * BP_BYTE_RANGE(17, 1) = 134217728 bytes
+ */
+#define BP_BYTE_RANGE(indblkshift, level) \
+ (BP_SPANB(indblkshift, level) * ((uint64_t)1 << indblkshift))
+
/*
* A block is a hole when it has either 1) never been written to, or
* 2) is zero-filled. In both cases, ZFS can return all zeroes for all reads
diff --git a/sys/contrib/openzfs/include/sys/zfs_stat.h b/sys/contrib/openzfs/include/sys/zfs_stat.h
index 312c45188134..33b68fc1f2b4 100644
--- a/sys/contrib/openzfs/include/sys/zfs_stat.h
+++ b/sys/contrib/openzfs/include/sys/zfs_stat.h
@@ -38,7 +38,28 @@ typedef struct zfs_stat {
} zfs_stat_t;
extern int zfs_obj_to_stats(objset_t *osp, uint64_t obj, zfs_stat_t *sb,
- char *buf, int len);
+ char *buf, int len, nvlist_t *nv);
+
+/*
+ * The legacy behavior of ZFS_IOC_OBJ_TO_STATS is to return a zfs_stat_t struct.
+ * However, if the user passes in a nvlist dst buffer, we also return
+ * "extended" object stats. Currently, these extended stats are handpicked
+ * fields from dmu_object_info_t, but they could be expanded to include
+ * anything.
+ */
+#define ZFS_OBJ_STAT_DATA_BLOCK_SIZE "data_block_size"
+#define ZFS_OBJ_STAT_METADATA_BLOCK_SIZE "metadata_block_size"
+#define ZFS_OBJ_STAT_DNODE_SIZE "dnode_size"
+#define ZFS_OBJ_STAT_TYPE "type"
+#define ZFS_OBJ_STAT_TYPE_STR "type_str"
+#define ZFS_OBJ_STAT_BONUS_TYPE "bonus_type"
+#define ZFS_OBJ_STAT_BONUS_TYPE_STR "bonus_type_str"
+#define ZFS_OBJ_STAT_BONUS_SIZE "bonus_size"
+#define ZFS_OBJ_STAT_CHECKSUM "checksum"
+#define ZFS_OBJ_STAT_COMPRESS "compress"
+#define ZFS_OBJ_STAT_PHYSICAL_BLOCKS_512 "physical_blocks_512"
+#define ZFS_OBJ_STAT_MAX_OFFSET "max_offset"
+#define ZFS_OBJ_STAT_FILL_COUNT "fill_count"
#ifdef __cplusplus
}
diff --git a/sys/contrib/openzfs/lib/libzfs/libzfs_pool.c b/sys/contrib/openzfs/lib/libzfs/libzfs_pool.c
index 9a26e88eb821..b69c1389330b 100644
--- a/sys/contrib/openzfs/lib/libzfs/libzfs_pool.c
+++ b/sys/contrib/openzfs/lib/libzfs/libzfs_pool.c
@@ -35,6 +35,7 @@
#include <zone.h>
#include <sys/stat.h>
#include <sys/efi_partition.h>
+#include <sys/range_tree.h>
#include <sys/systeminfo.h>
#include <sys/zfs_ioctl.h>
#include <sys/zfs_sysfs.h>
@@ -51,6 +52,11 @@
#include "zfeature_common.h"
static boolean_t zpool_vdev_is_interior(const char *name);
+static nvlist_t *zpool_get_extended_objset_stat(zpool_handle_t *zhp,
+ uint64_t dsobj, uint64_t obj);
+
+static nvlist_t *
+zpool_get_extended_obj_stat(zpool_handle_t *zhp, uint64_t dsobj, uint64_t obj);
typedef struct prop_flags {
unsigned int create:1; /* Validate property on creation */
@@ -4889,6 +4895,201 @@ zpool_add_propname(zpool_handle_t *zhp, const char *propname)
zhp->zpool_n_propnames++;
}
+/*
+ * Given a properties nvlist like:
+ *
+ * refreservation:
+ * source: 'tank/vol'
+ * value: 1092616192
+ * recordsize:
+ * value: 4096
+ * source: 'tank'
+ * refcompressratio:
+ * value: 100
+ * logicalreferenced:
+ * value: 1076408320
+ * compressratio:
+ * value: 100
+ * ...
+ *
+ * Lookup the 'value' field for a uint64_t and return it into *val. For
+ * example, if you pass "recordsize" for the name, it will store 4096 into *val.
+ */
+static int
+zpool_get_from_prop_nvlist_uint64(nvlist_t *nv, const char *name, uint64_t *val)
+{
+ nvlist_t *tmp;
+ int rc;
+
+ rc = nvlist_lookup_nvlist(nv, name, &tmp);
+ if (rc != 0)
+ return (rc);
+
+ return (nvlist_lookup_uint64(tmp, "value", val));
+}
+
+/*
+ * Given a dataset object and object number, return its data blocks size
+ * and type string (type_str). type_str must be freed when no longer needed.
+ */
+static int
+zpool_get_extended_obj_stat_helper(zpool_handle_t *zhp, uint64_t dsobj,
+ uint64_t obj, uint64_t *data_block_size, char **type_str)
+{
+ nvlist_t *nv;
+ boolean_t is_zvol = B_FALSE;
+ uint64_t val;
+ int rc;
+
+ nv = zpool_get_extended_obj_stat(zhp, dsobj, obj);
+ if (nv == NULL) {
+ nv = zpool_get_extended_objset_stat(zhp, dsobj, obj);
+ if (nv == NULL) {
+ return (-1);
+ }
+ is_zvol = B_TRUE;
+ }
+ if (is_zvol) {
+ rc = zpool_get_from_prop_nvlist_uint64(nv,
+ zfs_prop_to_name(ZFS_PROP_VOLBLOCKSIZE), &val);
+ } else {
+ uint32_t val32 = 0;
+ rc = nvlist_lookup_uint32(nv, ZFS_OBJ_STAT_DATA_BLOCK_SIZE,
+ &val32);
+ val = val32;
+ }
+
+ if (rc != 0) {
+ nvlist_free(nv);
+ return (rc);
+ }
+ *data_block_size = val;
+
+ const char *tmp;
+ if (is_zvol)
+ tmp = "zvol";
+ else
+ tmp = fnvlist_lookup_string(nv, ZFS_OBJ_STAT_TYPE_STR);
+ *type_str = strdup(tmp);
+
+ nvlist_free(nv);
+ return (0);
+}
+
+static void zpool_get_errlog_process_file_cb(void *arg, uint64_t start,
+ uint64_t size) {
+ nvlist_t ***tmp = arg;
+ nvlist_t **nva_next = *tmp;
+ nvlist_t *nv;
+
+ nv = fnvlist_alloc();
+ fnvlist_add_uint64(nv, ZPOOL_ERR_START_BYTE, start);
+ fnvlist_add_uint64(nv, ZPOOL_ERR_END_BYTE, start + size - 1);
+ *nva_next = nv;
+
+ /* Advance to next array entry */
+ *tmp = nva_next + 1;
+}
+
+static void
+zpool_get_errlog_process_file(zpool_handle_t *zhp,
+ nvlist_t **nverrlistp, zbookmark_phys_t *zb_start, zbookmark_phys_t *zb_end)
+{
+ uint64_t data_block_size = 0;
+ char *type_str = NULL;
+ nvlist_t *nv, **nva, **nva_next;
+ uint64_t count = 0;
+
+ /* Make the linter happy */
+ VERIFY(zb_start != NULL);
+ VERIFY(zb_end != NULL);
+
+ nv = fnvlist_alloc();
+ fnvlist_add_uint64(nv, ZPOOL_ERR_DATASET, zb_start->zb_objset);
+ fnvlist_add_uint64(nv, ZPOOL_ERR_OBJECT, zb_start->zb_object);
+
+ if (zpool_get_extended_obj_stat_helper(zhp, zb_start->zb_objset,
+ zb_start->zb_object, &data_block_size, &type_str) != 0) {
+ /*
+ * If the kernel supports extended stats, then include them.
+ * If not, it's still OK.
+ */
+ goto end;
+ }
+
+ fnvlist_add_string(nv, ZPOOL_ERR_OBJECT_TYPE, type_str);
+ fnvlist_add_uint64(nv, ZPOOL_ERR_BLOCK_SIZE, data_block_size);
+ free(type_str);
+
+ zfs_range_tree_t *range_tree;
+ range_tree = zfs_range_tree_create(NULL, ZFS_RANGE_SEG64, NULL, 0, 0);
+ if (range_tree == NULL)
+ goto end;
+
+ do {
+ int data_block_size_shift;
+ uint64_t data_block_range;
+
+ /*
+ * If an L1 (or higher block) is damaged, it will affect a much
+ * larger byte range than a simple L0 block. Calculate these
+ * ranges correctly.
+ */
+ if (highbit64(data_block_size) != lowbit64(data_block_size)) {
+ /*
+ * Our data block size is not power-of-two. This is
+ * probably a single block file.
+ */
+ data_block_size_shift = 0;
+ data_block_range = data_block_size;
+ } else {
+ data_block_size_shift = highbit64(data_block_size) - 1;
+ data_block_range = BP_BYTE_RANGE(data_block_size_shift,
+ zb_start->zb_level);
+ }
+
+ /*
+ * The range we are adding could overlap other entries. Do
+ * a clear first to punch a hole in any existing ranges,
+ * then do the add. We have to do this since our add can't
+ * overlap an existing range.
+ */
+ zfs_range_tree_clear(range_tree,
+ zb_start->zb_blkid * data_block_range, data_block_range);
+ zfs_range_tree_add(range_tree,
+ zb_start->zb_blkid * data_block_range, data_block_range);
+
+ if (zb_start == zb_end)
+ break;
+ } while (zb_start++);
+
+ /*
+ * Our range tree has all our ranges. Construct an array of start/end
+ * entries.
+ */
+ count = zfs_range_tree_numsegs(range_tree);
+
+ nva = zfs_alloc(zhp->zpool_hdl, sizeof (nvlist_t *) * count);
+ nva_next = &nva[0];
+
+ zfs_range_tree_walk(range_tree, zpool_get_errlog_process_file_cb,
+ &nva_next);
+
+ zfs_range_tree_vacate(range_tree, NULL, NULL);
+ zfs_range_tree_destroy(range_tree);
+
+ fnvlist_add_nvlist_array(nv, ZPOOL_ERR_RANGES, (const nvlist_t **) nva,
+ count);
+
+ for (uint64_t i = 0; i < count; i++)
+ nvlist_free(nva[i]);
+ free(nva);
+
+end:
+ fnvlist_add_nvlist(*nverrlistp, "ejk", nv);
+ nvlist_free(nv);
+}
+
/*
* Retrieve the persistent error log, uniquify the members, and return to the
* caller.
@@ -4900,6 +5101,7 @@ zpool_get_errlog(zpool_handle_t *zhp, nvlist_t **nverrlistp)
libzfs_handle_t *hdl = zhp->zpool_hdl;
zbookmark_phys_t *buf;
uint64_t buflen = 10000; /* approx. 1MB of RAM */
+ uint64_t i;
if (fnvlist_lookup_uint64(zhp->zpool_config,
ZPOOL_CONFIG_ERRCOUNT) == 0)
@@ -4939,6 +5141,10 @@ zpool_get_errlog(zpool_handle_t *zhp, nvlist_t **nverrlistp)
*/
zbookmark_phys_t *zb = buf + zc.zc_nvlist_dst_size;
uint64_t zblen = buflen - zc.zc_nvlist_dst_size;
+ if (zblen == 0) {
+ free(buf);
+ return (0); /* nothing to do */
+ }
qsort(zb, zblen, sizeof (zbookmark_phys_t), zbookmark_mem_compare);
@@ -4947,39 +5153,35 @@ zpool_get_errlog(zpool_handle_t *zhp, nvlist_t **nverrlistp)
/*
* Fill in the nverrlistp with nvlist's of dataset and object numbers.
*/
- for (uint64_t i = 0; i < zblen; i++) {
- nvlist_t *nv;
+ zbookmark_phys_t *start = NULL;
- /* ignoring zb_blkid and zb_level for now */
- if (i > 0 && zb[i-1].zb_objset == zb[i].zb_objset &&
- zb[i-1].zb_object == zb[i].zb_object)
+ for (i = 0; i < zblen; i++) {
+ if (start == NULL) {
+ start = &zb[i];
continue;
-
- if (nvlist_alloc(&nv, NV_UNIQUE_NAME, KM_SLEEP) != 0)
- goto nomem;
- if (nvlist_add_uint64(nv, ZPOOL_ERR_DATASET,
- zb[i].zb_objset) != 0) {
- nvlist_free(nv);
- goto nomem;
}
- if (nvlist_add_uint64(nv, ZPOOL_ERR_OBJECT,
- zb[i].zb_object) != 0) {
- nvlist_free(nv);
- goto nomem;
- }
- if (nvlist_add_nvlist(*nverrlistp, "ejk", nv) != 0) {
- nvlist_free(nv);
- goto nomem;
+
+ /* filter out duplicate files and levels */
+ if (zb[i-1].zb_objset == zb[i].zb_objset &&
+ zb[i-1].zb_object == zb[i].zb_object) {
+ /* same file, new error block */
+ continue;
+ } else {
+ /*
+ * Every time we see a new object, process the
+ * previous one.
+ */
+ zpool_get_errlog_process_file(zhp, nverrlistp,
+ start, &zb[i-1]);
+ start = &zb[i];
}
- nvlist_free(nv);
}
+ /* Process the last entry */
+ zpool_get_errlog_process_file(zhp, nverrlistp, start, &zb[i-1]);
free(buf);
- return (0);
-nomem:
- free(buf);
- return (no_memory(zhp->zpool_hdl));
+ return (0);
}
/*
@@ -5265,6 +5467,71 @@ zpool_events_seek(libzfs_handle_t *hdl, uint64_t eid, int zevent_fd)
return (error);
}
+/*
+ * Return extended information about an object. This calls the "extended"
+ * variant of ZFS_IOC_OBJ_TO_STATS to return things things like block size,
+ * dmu type, dnone size, etc (see dmu_object_info_t).
+ *
+ * Returned nvlist must be freed by the user when they are done with it.
+ */
+static nvlist_t *
+zpool_get_extended_obj_stat_impl(zpool_handle_t *zhp, uint64_t dsobj,
+ uint64_t obj, enum zfs_ioc stats_ioctl)
+{
+ zfs_cmd_t zc = {"\0"};
+ char dsname[ZFS_MAX_DATASET_NAME_LEN];
+ nvlist_t *nv = NULL;
+ int error;
+
+ /* get the dataset's name */
+ zc.zc_obj = dsobj;
+ (void) strlcpy(zc.zc_name, zhp->zpool_name, sizeof (zc.zc_name));
+ error = zfs_ioctl(zhp->zpool_hdl, ZFS_IOC_DSOBJ_TO_DSNAME, &zc);
+ if (error)
+ return (NULL);
+
+ (void) strlcpy(dsname, zc.zc_value, sizeof (dsname));
+ (void) strlcpy(zc.zc_name, dsname, sizeof (zc.zc_name));
+
+ zcmd_alloc_dst_nvlist(zhp->zpool_hdl, &zc, 1024);
+ zc.zc_obj = obj;
+ while (zfs_ioctl(zhp->zpool_hdl, stats_ioctl, &zc) != 0) {
+ if (errno == ENOMEM) {
+ zcmd_expand_dst_nvlist(zhp->zpool_hdl, &zc);
+ } else {
+ zcmd_free_nvlists(&zc);
+ return (NULL);
+ }
+ }
+
+ zcmd_read_dst_nvlist(zhp->zpool_hdl, &zc, &nv);
+ zcmd_free_nvlists(&zc);
+
+ return (nv);
+}
+
+/*
+ * Return extended information about an object. This calls the "extended"
+ * variant of ZFS_IOC_OBJ_TO_STATS to return things things like block size,
+ * dmu type, dnone size, etc (see dmu_object_info_t).
+ *
+ * Returned nvlist must be freed by the user when they are done with it.
+ */
+static nvlist_t *
+zpool_get_extended_obj_stat(zpool_handle_t *zhp, uint64_t dsobj, uint64_t obj)
+{
+ return (zpool_get_extended_obj_stat_impl(zhp, dsobj, obj,
+ ZFS_IOC_OBJ_TO_STATS));
+}
+
+static nvlist_t *
+zpool_get_extended_objset_stat(zpool_handle_t *zhp, uint64_t dsobj,
+ uint64_t obj)
+{
*** 466 LINES SKIPPED ***