Print this page
NEX-7298 powertop dumps core when -c, -d or -t flags are used
Reviewed by: Gordon Ross <gordon.ross@nexenta.com>
NEX-6725 zpool offlining/onlining first disk in a mirror causes checksum error
Reviewed by: Alek Pinchuk <alek.pinchuk@nexenta.com>
Reviewed by: Roman Strashkin <roman.strashkin@nexenta.com>
Reviewed by: Sanjay Nadkarni <sanjay.nadkarni@nexenta.com>
NEX-5795 Rename 'wrc' as 'wbc' in the source and in the tech docs
Reviewed by: Alex Aizman <alex.aizman@nexenta.com>
Reviewed by: Sanjay Nadkarni <sanjay.nadkarni@nexenta.com>
Reviewed by: Alek Pinchuk <alek.pinchuk@nexenta.com>
NEX-4683 WRC: Special block pointer must know that it is special
Reviewed by: Alex Aizman <alex.aizman@nexenta.com>
Reviewed by: Alek Pinchuk <alek.pinchuk@nexenta.com>
Reviewed by: Saso Kiselkov <saso.kiselkov@nexenta.com>
NEX-4245 WRC: Code cleanup and refactoring to simplify merge with upstream
Reviewed by: Alek Pinchuk <alek.pinchuk@nexenta.com>
Reviewed by: Alex Aizman <alex.aizman@nexenta.com>
NEX-4091 WRC: Scrub sometimes reports checksum error
Reviewed by: Alek Pinchuk <alek.pinchuk@nexenta.com>
Reviewed by: Saso Kiselkov <saso.kiselkov@nexenta.com>
NEX-3558 KRRP Integration
NEX-3508 CLONE - Port NEX-2946 Add UNMAP/TRIM functionality to ZFS and illumos
Reviewed by: Josef Sipek <josef.sipek@nexenta.com>
Reviewed by: Alek Pinchuk <alek.pinchuk@nexenta.com>
Conflicts:
    usr/src/uts/common/io/scsi/targets/sd.c
    usr/src/uts/common/sys/scsi/targets/sddef.h
NEX-1007 added checks for NULL vdev in mirror_map
NEX-801 If a block pointer is corrupt read or write may crash
If block pointer is corrupt in such a way that vdev id of one of the
ditto blocks is wrong (out of range), zio_vdev_io_start or zio_vdev_io_done
may trip over it and crash.
This changeset takes care of this by claiming that an invalid vdev is
neither readable, nor writeable.
OS-80 support for vdev and CoS properties for the new I/O scheduler
OS-95 lint warning introduced by OS-61
Moved closed ZFS files to open repo, changed Makefiles accordingly
Removed unneeded weak symbols
re #12585 rb4049 ZFS++ work port - refactoring to improve separation of open/closed code, bug fixes, performance improvements - open code
Bug 11205: add missing libzfs_closed_stubs.c to fix opensource-only build.
ZFS plus work: special vdevs, cos, cos/vdev properties

*** 22,31 **** --- 22,32 ---- * Copyright 2010 Sun Microsystems, Inc. All rights reserved. * Use is subject to license terms. */ /* + * Copyright 2016 Nexenta Systems, Inc. All rights reserved. * Copyright (c) 2012, 2015 by Delphix. All rights reserved. */ #include <sys/zfs_context.h> #include <sys/spa.h>
*** 33,42 **** --- 34,44 ---- #include <sys/dsl_pool.h> #include <sys/dsl_scan.h> #include <sys/vdev_impl.h> #include <sys/zio.h> #include <sys/abd.h> + #include <sys/wbc.h> #include <sys/fs/zfs.h> /* * Virtual device vector for mirroring. */
*** 46,55 **** --- 48,60 ---- uint64_t mc_offset; int mc_error; uint8_t mc_tried; uint8_t mc_skipped; uint8_t mc_speculative; + int mc_index; /* index in mirror_map_t */ + avl_node_t mc_node; /* used for sorting based on weight */ + int64_t mc_weight; /* thread-local copy of vdev_weight */ } mirror_child_t; typedef struct mirror_map { int mm_children; int mm_resilvering;
*** 82,117 **** int c, d; if (vd == NULL) { dva_t *dva = zio->io_bp->blk_dva; spa_t *spa = zio->io_spa; - dva_t dva_copy[SPA_DVAS_PER_BP]; c = BP_GET_NDVAS(zio->io_bp); - /* - * If we do not trust the pool config, some DVAs might be - * invalid or point to vdevs that do not exist. We skip them. - */ - if (!spa_trust_config(spa)) { - ASSERT3U(zio->io_type, ==, ZIO_TYPE_READ); - int j = 0; - for (int i = 0; i < c; i++) { - if (zfs_dva_valid(spa, &dva[i], zio->io_bp)) - dva_copy[j++] = dva[i]; - } - if (j == 0) { - zio->io_vsd = NULL; - zio->io_error = ENXIO; - return (NULL); - } - if (j < c) { - dva = dva_copy; - c = j; - } - } - mm = kmem_zalloc(offsetof(mirror_map_t, mm_child[c]), KM_SLEEP); mm->mm_children = c; mm->mm_resilvering = B_FALSE; mm->mm_preferred = spa_get_random(c); mm->mm_root = B_TRUE; --- 87,99 ----
*** 131,140 **** --- 113,125 ---- for (c = 0; c < mm->mm_children; c++) { mc = &mm->mm_child[c]; mc->mc_vd = vdev_lookup_top(spa, DVA_GET_VDEV(&dva[c])); mc->mc_offset = DVA_GET_OFFSET(&dva[c]); + mc->mc_index = c; + mc->mc_weight = (mc->mc_vd != NULL ? + mc->mc_vd->vdev_weight : 0); } } else { int replacing; c = vd->vdev_children;
*** 185,194 **** --- 170,182 ---- for (c = 0; c < mm->mm_children; c++) { mc = &mm->mm_child[c]; mc->mc_vd = vd->vdev_child[c]; mc->mc_offset = zio->io_offset; + mc->mc_index = c; + mc->mc_weight = (mc->mc_vd != NULL ? + mc->mc_vd->vdev_weight : 0); } } zio->io_vsd = mm; zio->io_vsd_ops = &vdev_mirror_vsd_ops;
*** 222,234 **** *max_asize = MIN(*max_asize - 1, cvd->vdev_max_asize - 1) + 1; *ashift = MAX(*ashift, cvd->vdev_ashift); } if (numerrors == vd->vdev_children) { - if (vdev_children_are_offline(vd)) - vd->vdev_stat.vs_aux = VDEV_AUX_CHILDREN_OFFLINE; - else vd->vdev_stat.vs_aux = VDEV_AUX_NO_REPLICAS; return (lasterror); } return (0); --- 210,219 ----
*** 274,327 **** mc->mc_error = zio->io_error; mc->mc_tried = 1; mc->mc_skipped = 0; } /* * Try to find a child whose DTL doesn't contain the block we want to read. * If we can't, try the read on any vdev we haven't already tried. */ static int vdev_mirror_child_select(zio_t *zio) { mirror_map_t *mm = zio->io_vsd; - mirror_child_t *mc; uint64_t txg = zio->io_txg; ! int i, c; ASSERT(zio->io_bp == NULL || BP_PHYSICAL_BIRTH(zio->io_bp) == txg); /* ! * Try to find a child whose DTL doesn't contain the block to read. ! * If a child is known to be completely inaccessible (indicated by ! * vdev_readable() returning B_FALSE), don't even try. */ ! for (i = 0, c = mm->mm_preferred; i < mm->mm_children; i++, c++) { ! if (c >= mm->mm_children) ! c = 0; ! mc = &mm->mm_child[c]; ! if (mc->mc_tried || mc->mc_skipped) continue; ! if (!vdev_readable(mc->mc_vd)) { ! mc->mc_error = SET_ERROR(ENXIO); ! mc->mc_tried = 1; /* don't even try */ ! mc->mc_skipped = 1; ! continue; } - if (!vdev_dtl_contains(mc->mc_vd, DTL_MISSING, txg, 1)) - return (c); - mc->mc_error = SET_ERROR(ESTALE); - mc->mc_skipped = 1; - mc->mc_speculative = 1; } /* * Every device is either missing or has this txg in its DTL. * Look for any child we haven't already tried before giving up. */ ! for (c = 0; c < mm->mm_children; c++) ! if (!mm->mm_child[c].mc_tried) return (c); /* * Every child failed. There's no place left to look. */ --- 259,399 ---- mc->mc_error = zio->io_error; mc->mc_tried = 1; mc->mc_skipped = 0; } + static int + vdev_weight_compar(const void *mc_a, const void *mc_b) + { + const mirror_child_t *a = mc_a, *b = mc_b; + + /* + * 1) if a's weight is less than b's, a goes right in the tree + * 2) if a's weight is greater than b's, a goes left + * 3) if a's and b's weights are equal, lower map index goes left + * 4) if weight and map index are equal, it's the same object + */ + if (a->mc_weight < b->mc_weight) + return (1); + if (a->mc_weight > b->mc_weight) + return (-1); + if (a->mc_index > b->mc_index) + return (1); + if (a->mc_index < b->mc_index) + return (-1); + ASSERT3P(a->mc_vd, ==, b->mc_vd); + return (0); + } + + static boolean_t + child_select_mc(mirror_child_t *mc, uint64_t txg) + { + if (mc->mc_tried || mc->mc_skipped) + return (B_FALSE); + if (!vdev_readable(mc->mc_vd)) { + mc->mc_error = SET_ERROR(ENXIO); + mc->mc_tried = 1; /* don't even try */ + mc->mc_skipped = 1; + return (B_FALSE); + } + if (!vdev_dtl_contains(mc->mc_vd, DTL_MISSING, txg, 1)) { + mc->mc_weight--; + mc->mc_vd->vdev_weight = mc->mc_weight; + return (B_TRUE); + } + mc->mc_error = SET_ERROR(ESTALE); + mc->mc_skipped = 1; + mc->mc_speculative = 1; + return (B_FALSE); + } + + static void + child_select_cleanup(mirror_map_t *mm, avl_tree_t *vdevs_by_weight) + { + for (int i = 0; i < mm->mm_children; i++) + avl_remove(vdevs_by_weight, &mm->mm_child[i]); + avl_destroy(vdevs_by_weight); + } + /* * Try to find a child whose DTL doesn't contain the block we want to read. * If we can't, try the read on any vdev we haven't already tried. */ static int vdev_mirror_child_select(zio_t *zio) { mirror_map_t *mm = zio->io_vsd; uint64_t txg = zio->io_txg; ! /* ! * Look at the weights of the vdevs in the mirror; the weights help ! * decide which vdev to read from; the highest-weight suitable child ! * index is returned, and its weight is decremented in order to avoid ! * creating "hot" devices; once all the vdevs' weights are zero, the ! * weights are set back to the ones configured in vdev props ! */ ! int64_t max_weight = 0; ASSERT(zio->io_bp == NULL || BP_PHYSICAL_BIRTH(zio->io_bp) == txg); + for (int c = 0; c < mm->mm_children; c++) { + mirror_child_t *mc = &mm->mm_child[c]; + if (mc->mc_vd == NULL) + continue; + max_weight = MAX(max_weight, mc->mc_weight); + } + /* ! * Recalculate weights */ ! if (max_weight == 0) { ! for (int c = 0; c < mm->mm_children; c++) { ! mirror_child_t *mc = &mm->mm_child[c]; ! if (mc->mc_vd == NULL) continue; ! mc->mc_weight = ! vdev_queue_get_prop_uint64(&mc->mc_vd->vdev_queue, ! VDEV_PROP_PREFERRED_READ) + 1; ! mc->mc_vd->vdev_weight = mc->mc_weight; } } + if (mm->mm_children > 1) { + avl_tree_t vdevs_by_weight; + + avl_create(&vdevs_by_weight, vdev_weight_compar, + sizeof (mirror_child_t), offsetof(mirror_child_t, mc_node)); + /* + * Sort the weighted list + */ + for (int i = 0; i < mm->mm_children; i++) + avl_add(&vdevs_by_weight, &mm->mm_child[i]); + + /* + * Try to find a child whose DTL doesn't contain the block to + * read. If a child is known to be completely inaccessible + * (vdev_readable() returning B_FALSE), don't even try. + */ + for (mirror_child_t *mc = avl_first(&vdevs_by_weight); + mc != NULL; mc = AVL_NEXT(&vdevs_by_weight, mc)) { + if (child_select_mc(mc, txg)) { + child_select_cleanup(mm, &vdevs_by_weight); + return (mc->mc_index); + } + } + child_select_cleanup(mm, &vdevs_by_weight); + } else { + if (child_select_mc(&mm->mm_child[0], txg)) + return (0); + } + + /* * Every device is either missing or has this txg in its DTL. * Look for any child we haven't already tried before giving up. */ ! for (int c = 0; c < mm->mm_children; c++) ! if (!mm->mm_child[c].mc_tried && mm->mm_child[c].mc_vd != NULL) return (c); /* * Every child failed. There's no place left to look. */
*** 332,361 **** vdev_mirror_io_start(zio_t *zio) { mirror_map_t *mm; mirror_child_t *mc; int c, children; mm = vdev_mirror_map_alloc(zio); ! if (mm == NULL) { ! ASSERT(!spa_trust_config(zio->io_spa)); ! ASSERT(zio->io_type == ZIO_TYPE_READ); ! zio_execute(zio); ! return; ! } if (zio->io_type == ZIO_TYPE_READ) { if ((zio->io_flags & ZIO_FLAG_SCRUB) && !mm->mm_resilvering) { /* * For scrubbing reads we need to allocate a read * buffer for each child and issue reads to all * children. If any child succeeds, it will copy its * data into zio->io_data in vdev_mirror_scrub_done. */ for (c = 0; c < mm->mm_children; c++) { mc = &mm->mm_child[c]; zio_nowait(zio_vdev_child_io(zio, zio->io_bp, mc->mc_vd, mc->mc_offset, abd_alloc_sametype(zio->io_abd, zio->io_size), zio->io_size, zio->io_type, zio->io_priority, 0, --- 404,455 ---- vdev_mirror_io_start(zio_t *zio) { mirror_map_t *mm; mirror_child_t *mc; int c, children; + boolean_t spec_case = B_FALSE; + spa_t *spa = zio->io_spa; mm = vdev_mirror_map_alloc(zio); ! if (zio->io_child_type != ZIO_CHILD_VDEV && ! BP_IS_SPECIAL(zio->io_bp)) ! spec_case = B_TRUE; if (zio->io_type == ZIO_TYPE_READ) { if ((zio->io_flags & ZIO_FLAG_SCRUB) && !mm->mm_resilvering) { + int target = 0; + /* + * Scrub of special BPs should take into + * account the state of WBC-Window + */ + if (spec_case) { + target = wbc_select_dva( + spa_get_wbc_data(spa), zio); + } + + /* * For scrubbing reads we need to allocate a read * buffer for each child and issue reads to all * children. If any child succeeds, it will copy its * data into zio->io_data in vdev_mirror_scrub_done. */ for (c = 0; c < mm->mm_children; c++) { mc = &mm->mm_child[c]; + if (mc->mc_vd == NULL) { + /* + * Invalid vdev id in blkptr caused + * mc_vd to be NULL here. + * Just skip this vdev. + */ + continue; + } + + if (spec_case && c != target) + continue; + zio_nowait(zio_vdev_child_io(zio, zio->io_bp, mc->mc_vd, mc->mc_offset, abd_alloc_sametype(zio->io_abd, zio->io_size), zio->io_size, zio->io_type, zio->io_priority, 0,
*** 365,375 **** --- 459,474 ---- return; } /* * For normal reads just pick one child. */ + + if (spec_case) + c = wbc_select_dva(spa_get_wbc_data(spa), zio); + else c = vdev_mirror_child_select(zio); + children = (c >= 0); } else { ASSERT(zio->io_type == ZIO_TYPE_WRITE); /*
*** 377,393 **** */ c = 0; children = mm->mm_children; } ! while (children--) { mc = &mm->mm_child[c]; zio_nowait(zio_vdev_child_io(zio, zio->io_bp, mc->mc_vd, mc->mc_offset, zio->io_abd, zio->io_size, zio->io_type, zio->io_priority, 0, vdev_mirror_child_done, mc)); ! c++; } zio_execute(zio); } --- 476,508 ---- */ c = 0; children = mm->mm_children; } ! for (; children--; c++) { mc = &mm->mm_child[c]; + if (mc->mc_vd == NULL) { + /* + * Invalid vdev in blkptr caused mc_vd to be NULL here. + * Just skip this vdev. + */ + continue; + } + + if (spec_case) { + if (zio->io_type == ZIO_TYPE_WRITE && + !vdev_is_special(mc->mc_vd)) + continue; + } + zio_nowait(zio_vdev_child_io(zio, zio->io_bp, mc->mc_vd, mc->mc_offset, zio->io_abd, zio->io_size, zio->io_type, zio->io_priority, 0, vdev_mirror_child_done, mc)); ! ! if (spec_case) ! break; } zio_execute(zio); }
*** 412,424 **** mirror_child_t *mc; int c; int good_copies = 0; int unexpected_errors = 0; - if (mm == NULL) - return; - for (c = 0; c < mm->mm_children; c++) { mc = &mm->mm_child[c]; if (mc->mc_error) { if (!mc->mc_skipped) --- 527,536 ----
*** 520,542 **** } static void vdev_mirror_state_change(vdev_t *vd, int faulted, int degraded) { ! if (faulted == vd->vdev_children) { ! if (vdev_children_are_offline(vd)) { ! vdev_set_state(vd, B_FALSE, VDEV_STATE_OFFLINE, ! VDEV_AUX_CHILDREN_OFFLINE); ! } else { vdev_set_state(vd, B_FALSE, VDEV_STATE_CANT_OPEN, VDEV_AUX_NO_REPLICAS); ! } ! } else if (degraded + faulted != 0) { vdev_set_state(vd, B_FALSE, VDEV_STATE_DEGRADED, VDEV_AUX_NONE); ! } else { vdev_set_state(vd, B_FALSE, VDEV_STATE_HEALTHY, VDEV_AUX_NONE); - } } vdev_ops_t vdev_mirror_ops = { vdev_mirror_open, vdev_mirror_close, --- 632,648 ---- } static void vdev_mirror_state_change(vdev_t *vd, int faulted, int degraded) { ! if (faulted == vd->vdev_children) vdev_set_state(vd, B_FALSE, VDEV_STATE_CANT_OPEN, VDEV_AUX_NO_REPLICAS); ! else if (degraded + faulted != 0) vdev_set_state(vd, B_FALSE, VDEV_STATE_DEGRADED, VDEV_AUX_NONE); ! else vdev_set_state(vd, B_FALSE, VDEV_STATE_HEALTHY, VDEV_AUX_NONE); } vdev_ops_t vdev_mirror_ops = { vdev_mirror_open, vdev_mirror_close,