/* *Findthedisknumberwhichtriggeredgivenbio
*/ staticint find_bio_disk(struct r10conf *conf, struct r10bio *r10_bio, struct bio *bio, int *slotp, int *replp)
{ int slot; int repl = 0;
for (slot = 0; slot < conf->geo.raid_disks; slot++) { if (r10_bio->devs[slot].bio == bio) break; if (r10_bio->devs[slot].repl_bio == bio) {
repl = 1; break;
}
}
update_head_pos(slot, r10_bio);
if (slotp)
*slotp = slot; if (replp)
*replp = repl; return r10_bio->devs[slot].devnum;
}
staticvoid raid10_end_read_request(struct bio *bio)
{ int uptodate = !bio->bi_status; struct r10bio *r10_bio = bio->bi_private; int slot; struct md_rdev *rdev; struct r10conf *conf = r10_bio->mddev->private;
staticvoid one_write_done(struct r10bio *r10_bio)
{ if (atomic_dec_and_test(&r10_bio->remaining)) { if (test_bit(R10BIO_WriteError, &r10_bio->state))
reschedule_retry(r10_bio); else {
close_write(r10_bio); if (test_bit(R10BIO_MadeGood, &r10_bio->state))
reschedule_retry(r10_bio); else
raid_end_bio_io(r10_bio);
}
}
}
staticvoid raid10_end_write_request(struct bio *bio)
{ struct r10bio *r10_bio = bio->bi_private; int dev; int dec_rdev = 1; struct r10conf *conf = r10_bio->mddev->private; int slot, repl; struct md_rdev *rdev = NULL; struct bio *to_put = NULL; bool ignore_error = !raid1_should_handle_error(bio) ||
(bio->bi_status && bio_op(bio) == REQ_OP_DISCARD);
dev = find_bio_disk(conf, r10_bio, bio, &slot, &repl);
if (repl)
rdev = conf->mirrors[dev].replacement; if (!rdev) {
smp_rmb();
repl = 0;
rdev = conf->mirrors[dev].rdev;
} /* *thisbranchisour'onemirrorIOhasfinished'eventhandler:
*/ if (bio->bi_status && !ignore_error) { if (repl) /* Never record new bad blocks to replacement, *justfailit.
*/
md_error(rdev->mddev, rdev); else {
set_bit(WriteErrorSeen, &rdev->flags); if (!test_and_set_bit(WantReplacement, &rdev->flags))
set_bit(MD_RECOVERY_NEEDED,
&rdev->mddev->recovery);
/* now calculate first sector/dev */
chunk = r10bio->sector >> geo->chunk_shift;
sector = r10bio->sector & geo->chunk_mask;
chunk *= geo->near_copies;
stripe = chunk;
dev = sector_div(stripe, geo->raid_disks); if (geo->far_offset)
stripe *= geo->far_copies;
sector += stripe << geo->chunk_shift;
/* and calculate all the others */ for (n = 0; n < geo->near_copies; n++) { int d = dev; int set;
sector_t s = sector;
r10bio->devs[slot].devnum = d;
r10bio->devs[slot].addr = s;
slot++;
for (f = 1; f < geo->far_copies; f++) {
set = d / geo->far_set_size;
d += geo->near_copies;
if ((geo->raid_disks % geo->far_set_size) &&
(d > last_far_set_start)) {
d -= last_far_set_start;
d %= last_far_set_size;
d += last_far_set_start;
} else {
d %= geo->far_set_size;
d += geo->far_set_size * set;
}
s += geo->stride;
r10bio->devs[slot].devnum = d;
r10bio->devs[slot].addr = s;
slot++;
}
dev++; if (dev >= geo->raid_disks) {
dev = 0;
sector += (geo->chunk_mask + 1);
}
}
}
static sector_t raid10_find_virt(struct r10conf *conf, sector_t sector, int dev)
{
sector_t offset, chunk, vchunk; /* Never use conf->prev as this is only called during resync *orrecovery,soreshapeisn'thappening
*/ struct geom *geo = &conf->geo; int far_set_start = (dev / geo->far_set_size) * geo->far_set_size; int far_set_size = geo->far_set_size; int last_far_set_start;
if (best_dist_slot >= 0) /* At least 2 disks to choose from so failfast is OK */
set_bit(R10BIO_FailFast, &r10_bio->state); /* This optimisation is debatable, and completely destroys *sequentialreadspeedfor'farcopies'arrays.Soonly *keepitfor'near'arrays,andreviewthoselater.
*/ if (geo->near_copies > 1 && !pending)
new_distance = 0;
/* for far > 1 always use the lowest address */ elseif (geo->far_copies > 1)
new_distance = r10_bio->devs[slot].addr; else
new_distance = abs(r10_bio->devs[slot].addr -
conf->mirrors[disk].head_position);
staticvoid flush_pending_writes(struct r10conf *conf)
{ /* Any writes that have been queued but are awaiting *bitmapupdatesgetflushedhere.
*/
spin_lock_irq(&conf->device_lock);
if (conf->pending_bio_list.head) { struct blk_plug plug; struct bio *bio;
bio = bio_list_get(&conf->pending_bio_list);
spin_unlock_irq(&conf->device_lock);
write_seqlock_irq(&conf->resync_lock); if (conf->barrier) { /* Return false when nowait flag is set */ if (nowait) {
ret = false;
} else {
conf->nr_waiting++;
mddev_add_trace_msg(conf->mddev, "raid10 wait barrier");
wait_event_barrier(conf, stop_waiting_barrier(conf));
conf->nr_waiting--;
} if (!conf->nr_waiting)
wake_up(&conf->wait_barrier);
} /* Only increment nr_pending when we wait */ if (ret)
atomic_inc(&conf->nr_pending);
write_sequnlock_irq(&conf->resync_lock); return ret;
}
/* we aren't scheduling, so we can do the write-out directly. */
bio = bio_list_get(&plug->pending);
raid1_prepare_flush_writes(mddev);
wake_up_barrier(conf);
while (bio) { /* submit pending writes */ struct bio *next = bio->bi_next;
raid1_submit_write(bio);
bio = next;
cond_resched();
}
kfree(plug);
}
/* *1.Registerthenewrequestandwaitifthereconstructionthreadhasput *upabarfornewrequests.Continueimmediatelyifnoresyncisactive *currently. *2.IfIOspansthereshapeposition.Needtowaitforreshapetopass.
*/ staticbool regular_request_wait(struct mddev *mddev, struct r10conf *conf, struct bio *bio, sector_t sectors)
{ /* Bail out if REQ_NOWAIT is set for the bio */ if (!wait_barrier(conf, bio->bi_opf & REQ_NOWAIT)) {
bio_wouldblock_error(bio); returnfalse;
} while (test_bit(MD_RECOVERY_RESHAPE, &mddev->recovery) &&
bio->bi_iter.bi_sector < conf->reshape_progress &&
bio->bi_iter.bi_sector + sectors > conf->reshape_progress) {
allow_barrier(conf); if (bio->bi_opf & REQ_NOWAIT) {
bio_wouldblock_error(bio); returnfalse;
}
mddev_add_trace_msg(conf->mddev, "raid10 wait reshape");
wait_event(conf->wait_barrier,
conf->reshape_progress <= bio->bi_iter.bi_sector ||
conf->reshape_progress >= bio->bi_iter.bi_sector +
sectors);
wait_barrier(conf, false);
} returntrue;
}
staticvoid raid10_read_request(struct mddev *mddev, struct bio *bio, struct r10bio *r10_bio, bool io_accounting)
{ struct r10conf *conf = mddev->private; struct bio *read_bio; int max_sectors; struct md_rdev *rdev; char b[BDEVNAME_SIZE]; int slot = r10_bio->read_slot; struct md_rdev *err_rdev = NULL;
gfp_t gfp = GFP_NOIO; int error;
if (unlikely(blocked_rdev)) { /* Have to wait for this device to get unblocked, then retry */
allow_barrier(conf);
mddev_add_trace_msg(conf->mddev, "raid10 %s wait rdev %d blocked",
__func__, blocked_rdev->raid_disk);
md_wait_for_blocked_rdev(blocked_rdev, mddev);
wait_barrier(conf, false); goto retry_wait;
}
}
staticvoid raid10_write_request(struct mddev *mddev, struct bio *bio, struct r10bio *r10_bio)
{ struct r10conf *conf = mddev->private; int i, k;
sector_t sectors; int max_sectors; int error;
if ((mddev_is_clustered(mddev) &&
mddev->cluster_ops->area_resyncing(mddev, WRITE,
bio->bi_iter.bi_sector,
bio_end_sector(bio)))) {
DEFINE_WAIT(w); /* Bail out if REQ_NOWAIT is set for the bio */ if (bio->bi_opf & REQ_NOWAIT) {
bio_wouldblock_error(bio); return;
} for (;;) {
prepare_to_wait(&conf->wait_barrier,
&w, TASK_IDLE); if (!mddev->cluster_ops->area_resyncing(mddev, WRITE,
bio->bi_iter.bi_sector, bio_end_sector(bio))) break;
schedule();
}
finish_wait(&conf->wait_barrier, &w);
}
bio_chain(split, bio);
trace_block_split(split, bio->bi_iter.bi_sector);
allow_barrier(conf); /* Resend the second split part */
submit_bio_noacct(bio);
bio = split;
wait_barrier(conf, false);
}
/* check if there are enough drives for *everyblocktoappearonatleastone. *Don'tconsiderthedevicenumbered'ignore' *aswemightbeabouttoremoveit.
*/ staticint _enough(struct r10conf *conf, int previous, int ignore)
{ int first = 0; int has_enough = 0; int disks, ncopies; if (previous) {
disks = conf->prev.raid_disks;
ncopies = conf->prev.near_copies;
} else {
disks = conf->geo.raid_disks;
ncopies = conf->geo.near_copies;
}
do { int n = conf->copies; int cnt = 0; intthis = first; while (n--) { struct md_rdev *rdev; if (this != ignore &&
(rdev = conf->mirrors[this].rdev) &&
test_bit(In_sync, &rdev->flags))
cnt++; this = (this+1) % disks;
} if (cnt == 0) goto out;
first = (first + ncopies) % disks;
} while (first != 0);
has_enough = 1;
out: return has_enough;
}
staticint enough(struct r10conf *conf, int ignore)
{ /* when calling 'enough', both 'prev' and 'geo' must *bestable. *Thisisensuredif->reconfig_mutexor->device_lock *isheld.
*/ return _enough(conf, 0, ignore) &&
_enough(conf, 1, ignore);
}
lockdep_assert_held(&conf->mddev->reconfig_mutex); for (i = 0; i < conf->geo.raid_disks; i++) {
rdev = conf->mirrors[i].rdev; if (rdev)
pr_debug(" disk %d, wo:%d, o:%d, dev:%pg\n",
i, !test_bit(In_sync, &rdev->flags),
!test_bit(Faulty, &rdev->flags),
rdev->bdev);
}
}
/* *Findallnon-in_syncdiskswithintheRAID10configuration *andmarkthemin_sync
*/ for (i = 0; i < conf->geo.raid_disks; i++) {
tmp = conf->mirrors + i; if (tmp->replacement
&& tmp->replacement->recovery_offset == MaxSector
&& !test_bit(Faulty, &tmp->replacement->flags)
&& !test_and_set_bit(In_sync, &tmp->replacement->flags)) { /* Replacement has just become active */ if (!tmp->rdev
|| !test_and_clear_bit(In_sync, &tmp->rdev->flags))
count++; if (tmp->rdev) { /* Replaced device not technically faulty, *butweneedtobesureitgetsremoved *andneverre-added.
*/
set_bit(Faulty, &tmp->rdev->flags);
sysfs_notify_dirent_safe(
tmp->rdev->sysfs_state);
}
sysfs_notify_dirent_safe(tmp->replacement->sysfs_state);
} elseif (tmp->rdev
&& tmp->rdev->recovery_offset == MaxSector
&& !test_bit(Faulty, &tmp->rdev->flags)
&& !test_and_set_bit(In_sync, &tmp->rdev->flags)) {
count++;
sysfs_notify_dirent_safe(tmp->rdev->sysfs_state);
}
}
spin_lock_irqsave(&conf->device_lock, flags);
mddev->degraded -= count;
spin_unlock_irqrestore(&conf->device_lock, flags);
print_conf(conf); return count;
}
staticint raid10_add_disk(struct mddev *mddev, struct md_rdev *rdev)
{ struct r10conf *conf = mddev->private; int err = -EEXIST; int mirror, repl_slot = -1; int first = 0; int last = conf->geo.raid_disks - 1; struct raid10_info *p;
if (mddev->resync_offset < MaxSector) /* only hot-add to in-sync arrays, as recovery is *verydifferentfromresync
*/ return -EBUSY; if (rdev->saved_raid_disk < 0 && !_enough(conf, 1, -1)) return -EINVAL;
if (rdev->raid_disk >= 0)
first = last = rdev->raid_disk;
if (rdev->saved_raid_disk >= first &&
rdev->saved_raid_disk < conf->geo.raid_disks &&
conf->mirrors[rdev->saved_raid_disk].rdev == NULL)
mirror = rdev->saved_raid_disk; else
mirror = first; for ( ; mirror <= last ; mirror++) {
p = &conf->mirrors[mirror]; if (p->recovery_disabled == mddev->recovery_disabled) continue; if (p->rdev) { if (test_bit(WantReplacement, &p->rdev->flags) &&
p->replacement == NULL && repl_slot < 0)
repl_slot = mirror; continue;
}
staticvoid __end_sync_read(struct r10bio *r10_bio, struct bio *bio, int d)
{ struct r10conf *conf = r10_bio->mddev->private;
if (!bio->bi_status)
set_bit(R10BIO_Uptodate, &r10_bio->state); else /* The write handler will notice the lack of *R10BIO_Uptodateandrecordanyerrorsetc
*/
atomic_add(r10_bio->sectors,
&conf->mirrors[d].rdev->corrected_errors);
/* for reconstruct, we always reschedule after a read. *forresync,onlyafterallreads
*/
rdev_dec_pending(conf->mirrors[d].rdev, conf->mddev); if (test_bit(R10BIO_IsRecover, &r10_bio->state) ||
atomic_dec_and_test(&r10_bio->remaining)) { /* we have read all the blocks, *dothecomparisoninprocesscontextinraid10d
*/
reschedule_retry(r10_bio);
}
}
staticvoid end_sync_read(struct bio *bio)
{ struct r10bio *r10_bio = get_resync_r10bio(bio); struct r10conf *conf = r10_bio->mddev->private; int d = find_bio_disk(conf, r10_bio, bio, NULL, NULL);
__end_sync_read(r10_bio, bio, d);
}
staticvoid end_reshape_read(struct bio *bio)
{ /* reshape read bio isn't allocated from r10buf_pool */ struct r10bio *r10_bio = bio->bi_private;
vcnt = (r10_bio->sectors + (PAGE_SIZE >> 9) - 1) >> (PAGE_SHIFT - 9); /* now find blocks with errors */ for (i=0 ; i < conf->copies ; i++) { int j, d; struct md_rdev *rdev; struct resync_pages *rp;
tbio = r10_bio->devs[i].bio;
if (tbio->bi_end_io != end_sync_read) continue; if (i == first) continue;
tpages = get_resync_pages(tbio)->pages;
d = r10_bio->devs[i].devnum;
rdev = conf->mirrors[d].rdev; if (!r10_bio->devs[i].bio->bi_status) { /* We know that the bi_io_vec layout is the same for *both'first'and'i',sowejustcomparethem. *AllvecentriesarePAGE_SIZE;
*/ int sectors = r10_bio->sectors; for (j = 0; j < vcnt; j++) { int len = PAGE_SIZE; if (sectors < (len / 512))
len = sectors * 512; if (memcmp(page_address(fpages[j]),
page_address(tpages[j]),
len)) break;
sectors -= len/512;
} if (j == vcnt) continue;
atomic64_add(r10_bio->sectors, &mddev->resync_mismatches); if (test_bit(MD_RECOVERY_CHECK, &mddev->recovery)) /* Don't fix anything. */ continue;
} elseif (test_bit(FailFast, &rdev->flags)) { /* Just give up on this device */
md_error(rdev->mddev, rdev); continue;
} /* Ok, we need to write this bio, either to correct an *inconsistencyortocorrectanunreadableblock. *Firstweneedtofixupbv_offset,bv_lenand *bi_vecs,asthereadrequestmighthavecorruptedthese
*/
rp = get_resync_pages(tbio);
bio_reset(tbio, conf->mirrors[d].rdev->bdev, REQ_OP_WRITE);
/* Now write out to any replacement devices *thatareactive
*/ for (i = 0; i < conf->copies; i++) {
tbio = r10_bio->devs[i].repl_bio; if (!tbio || !tbio->bi_end_io) continue; if (r10_bio->devs[i].bio->bi_end_io != end_sync_write
&& r10_bio->devs[i].bio != fbio)
bio_copy_data(tbio, fbio);
atomic_inc(&r10_bio->remaining);
submit_bio_noacct(tbio);
}
done: if (atomic_dec_and_test(&r10_bio->remaining)) {
md_done_sync(mddev, r10_bio->sectors, 1);
put_buf(r10_bio);
}
}
/* *Nowfortherecoverycode. *Recoveryhappensacrossphysicalsectors. *Werecoverallnon-is_syncdrivesbyfindingthevirtualaddressof *each,andthenchooseaworkingdrivethatalsohasthatvirtaddress. *Thereisaseparater10_bioforeachnon-in_syncdrive. *Onlythefirsttwoslotsareinuse.Thefirstforreading, *Thesecondforwriting. *
*/ staticvoid fix_recovery_read_error(struct r10bio *r10_bio)
{ /* We got a read error during recovery. *Werepeatthereadinsmallerpage-sizedsections. *Ifareadsucceeds,writeittothenewdeviceorrecord *abadblockifwecannot. *Ifareadfails,recordabadblockonbotholdand *newdevices.
*/ struct mddev *mddev = r10_bio->mddev; struct r10conf *conf = mddev->private; struct bio *bio = r10_bio->devs[0].bio;
sector_t sect = 0; int sectors = r10_bio->sectors; int idx = 0; int dr = r10_bio->devs[0].devnum; int dw = r10_bio->devs[1].devnum; struct page **pages = get_resync_pages(bio)->pages;
while (sectors) { int s = sectors; struct md_rdev *rdev;
sector_t addr; int ok;
if (s > (PAGE_SIZE>>9))
s = PAGE_SIZE >> 9;
rdev = conf->mirrors[dr].rdev;
addr = r10_bio->devs[0].addr + sect;
ok = sync_page_io(rdev,
addr,
s << 9,
pages[idx],
REQ_OP_READ, false); if (ok) {
rdev = conf->mirrors[dw].rdev;
addr = r10_bio->devs[1].addr + sect;
ok = sync_page_io(rdev,
addr,
s << 9,
pages[idx],
REQ_OP_WRITE, false); if (!ok) {
set_bit(WriteErrorSeen, &rdev->flags); if (!test_and_set_bit(WantReplacement,
&rdev->flags))
set_bit(MD_RECOVERY_NEEDED,
&rdev->mddev->recovery);
}
} if (!ok) { /* We don't worry if we cannot set a bad block - *itreallyisbadsothereisnolossinnot *recordingityet
*/
rdev_set_badblocks(rdev, addr, s, 0);
if (rdev != conf->mirrors[dw].rdev) { /* need bad block on destination too */ struct md_rdev *rdev2 = conf->mirrors[dw].rdev;
addr = r10_bio->devs[1].addr + sect;
ok = rdev_set_badblocks(rdev2, addr, s, 0); if (!ok) { /* just abort the recovery */
pr_notice("md/raid10:%s: recovery aborted due to read error\n",
mdname(mddev));
staticvoid recovery_request_write(struct mddev *mddev, struct r10bio *r10_bio)
{ struct r10conf *conf = mddev->private; int d; struct bio *wbio = r10_bio->devs[1].bio; struct bio *wbio2 = r10_bio->devs[1].repl_bio;
/* Need to test wbio2->bi_end_io before we call *submit_bio_noacctasiftheformerisNULL, *thelatterisfreetofreewbio2.
*/ if (wbio2 && !wbio2->bi_end_io)
wbio2 = NULL;
if (!test_bit(R10BIO_Uptodate, &r10_bio->state)) {
fix_recovery_read_error(r10_bio); if (wbio->bi_end_io)
end_sync_request(r10_bio); if (wbio2)
end_sync_request(r10_bio); return;
}
/* *sharethepageswiththefirstbio *andsubmitthewriterequest
*/
d = r10_bio->devs[1].devnum; if (wbio->bi_end_io) {
atomic_inc(&conf->mirrors[d].rdev->nr_pending);
submit_bio_noacct(wbio);
} if (wbio2) {
atomic_inc(&conf->mirrors[d].replacement->nr_pending);
submit_bio_noacct(wbio2);
}
}
staticint r10_sync_page_io(struct md_rdev *rdev, sector_t sector, int sectors, struct page *page, enum req_op op)
{ if (rdev_has_badblock(rdev, sector, sectors) &&
(op == REQ_OP_READ || test_bit(WriteErrorSeen, &rdev->flags))) return -1; if (sync_page_io(rdev, sector, sectors << 9, page, op, false)) /* success */ return1; if (op == REQ_OP_WRITE) {
set_bit(WriteErrorSeen, &rdev->flags); if (!test_and_set_bit(WantReplacement, &rdev->flags))
set_bit(MD_RECOVERY_NEEDED,
&rdev->mddev->recovery);
} /* need to record an error - either for the block or the device */ if (!rdev_set_badblocks(rdev, sector, sectors, 0))
md_error(rdev->mddev, rdev); return0;
}
staticvoid fix_read_error(struct r10conf *conf, struct mddev *mddev, struct r10bio *r10_bio)
{ int sect = 0; /* Offset from r10_bio->sector */ int sectors = r10_bio->sectors, slot = r10_bio->read_slot; struct md_rdev *rdev; int d = r10_bio->devs[slot].devnum;
/* still own a reference to this rdev, so it cannot *havebeenclearedrecently.
*/
rdev = conf->mirrors[d].rdev;
if (test_bit(Faulty, &rdev->flags)) /* drive has already been failed, just ignore any
more fix_read_error() attempts */ return;
if (exceed_read_errors(mddev, rdev)) {
r10_bio->devs[slot].bio = IO_BLOCKED; return;
}
while(sectors) { int s = sectors; int sl = slot; int success = 0; int start;
if (s > (PAGE_SIZE>>9))
s = PAGE_SIZE >> 9;
do {
d = r10_bio->devs[sl].devnum;
rdev = conf->mirrors[d].rdev; if (rdev &&
test_bit(In_sync, &rdev->flags) &&
!test_bit(Faulty, &rdev->flags) &&
rdev_has_badblock(rdev,
r10_bio->devs[sl].addr + sect,
s) == 0) {
atomic_inc(&rdev->nr_pending);
success = sync_page_io(rdev,
r10_bio->devs[sl].addr +
sect,
s<<9,
conf->tmppage,
REQ_OP_READ, false);
rdev_dec_pending(rdev, mddev); if (success) break;
}
sl++; if (sl == conf->copies)
sl = 0;
} while (sl != slot);
if (!success) { /* Cannot read from anywhere, just mark the block *asbadonthefirstdevicetodiscouragefuture *reads.
*/ int dn = r10_bio->devs[slot].devnum;
rdev = conf->mirrors[dn].rdev;
start = sl; /* write it back and re-read */ while (sl != slot) { if (sl==0)
sl = conf->copies;
sl--;
d = r10_bio->devs[sl].devnum;
rdev = conf->mirrors[d].rdev; if (!rdev ||
test_bit(Faulty, &rdev->flags) ||
!test_bit(In_sync, &rdev->flags)) continue;
atomic_inc(&rdev->nr_pending); if (r10_sync_page_io(rdev,
r10_bio->devs[sl].addr +
sect,
s, conf->tmppage, REQ_OP_WRITE)
== 0) { /* Well, this device is dead */
pr_notice("md/raid10:%s: read correction write failed (%d sectors at %llu on %pg)\n",
mdname(mddev), s,
(unsignedlonglong)(
sect +
choose_data_offset(r10_bio,
rdev)),
rdev->bdev);
pr_notice("md/raid10:%s: %pg: failing drive\n",
mdname(mddev),
rdev->bdev);
}
rdev_dec_pending(rdev, mddev);
}
sl = start; while (sl != slot) { if (sl==0)
sl = conf->copies;
sl--;
d = r10_bio->devs[sl].devnum;
rdev = conf->mirrors[d].rdev; if (!rdev ||
test_bit(Faulty, &rdev->flags) ||
!test_bit(In_sync, &rdev->flags)) continue;
atomic_inc(&rdev->nr_pending); switch (r10_sync_page_io(rdev,
r10_bio->devs[sl].addr +
sect,
s, conf->tmppage, REQ_OP_READ)) { case0: /* Well, this device is dead */
pr_notice("md/raid10:%s: unable to read back corrected sectors (%d sectors at %llu on %pg)\n",
mdname(mddev), s,
(unsignedlonglong)(
sect +
choose_data_offset(r10_bio, rdev)),
rdev->bdev);
pr_notice("md/raid10:%s: %pg: failing drive\n",
mdname(mddev),
rdev->bdev); break; case1:
pr_info("md/raid10:%s: read error corrected (%d sectors at %llu on %pg)\n",
mdname(mddev), s,
(unsignedlonglong)(
sect +
choose_data_offset(r10_bio, rdev)),
rdev->bdev);
atomic_add(s, &rdev->corrected_errors);
}
rdev_dec_pending(rdev, mddev);
}
sectors -= s;
sect += s;
}
}
staticbool narrow_write_error(struct r10bio *r10_bio, int i)
{ struct bio *bio = r10_bio->master_bio; struct mddev *mddev = r10_bio->mddev; struct r10conf *conf = mddev->private; struct md_rdev *rdev = conf->mirrors[r10_bio->devs[i].devnum].rdev; /* bio has the data to be written to slot 'i' where *wejustrecentlyhadawriteerror. *Werepeatedlyclonethebioandtrimdowntooneblock, *thentrythewrite.Wherethewritefailswerecord *abadblock. *Itisconceivablethatthebiodoesn'texactlyalignwith *blocks.Wemusthandlethis. * *Wecurrentlyownareferencetotherdev.
*/
int block_sectors;
sector_t sector; int sectors; int sect_to_write = r10_bio->sectors; bool ok = true;
/* we got a read error. Maybe the drive is bad. Maybe just *theblockandwecanfixit. *WefreezeallotherIO,andtryreadingtheblockfrom *otherdevices.Whenwefindone,were-write *andcheckitthatfixesthereaderror. *Thisisalldonesynchronouslywhilethearrayis *frozen.
*/
bio = r10_bio->devs[slot].bio;
bio_put(bio);
r10_bio->devs[slot].bio = NULL;
staticvoid handle_write_completed(struct r10conf *conf, struct r10bio *r10_bio)
{ /* Some sort of write request has finished and it *succeededinwritingwherewethoughttherewasa *badblock.Soforgetthebadblock. *Orpossiblyiffailedandweneedtorecord *abadblock.
*/ int m; struct md_rdev *rdev;
if (test_bit(R10BIO_IsSync, &r10_bio->state) ||
test_bit(R10BIO_IsRecover, &r10_bio->state)) { for (m = 0; m < conf->copies; m++) { int dev = r10_bio->devs[m].devnum;
rdev = conf->mirrors[dev].rdev; if (r10_bio->devs[m].bio == NULL ||
r10_bio->devs[m].bio->bi_end_io == NULL) continue; if (!r10_bio->devs[m].bio->bi_status) {
rdev_clear_badblocks(
rdev,
r10_bio->devs[m].addr,
r10_bio->sectors, 0);
} else { if (!rdev_set_badblocks(
rdev,
r10_bio->devs[m].addr,
r10_bio->sectors, 0))
md_error(conf->mddev, rdev);
}
rdev = conf->mirrors[dev].replacement; if (r10_bio->devs[m].repl_bio == NULL ||
r10_bio->devs[m].repl_bio->bi_end_io == NULL) continue;
if (!r10_bio->devs[m].repl_bio->bi_status) {
rdev_clear_badblocks(
rdev,
r10_bio->devs[m].addr,
r10_bio->sectors, 0);
} else { if (!rdev_set_badblocks(
rdev,
r10_bio->devs[m].addr,
r10_bio->sectors, 0))
md_error(conf->mddev, rdev);
}
}
put_buf(r10_bio);
} else { bool fail = false; for (m = 0; m < conf->copies; m++) { int dev = r10_bio->devs[m].devnum; struct bio *bio = r10_bio->devs[m].bio;
rdev = conf->mirrors[dev].rdev; if (bio == IO_MADE_GOOD) {
rdev_clear_badblocks(
rdev,
r10_bio->devs[m].addr,
r10_bio->sectors, 0);
rdev_dec_pending(rdev, conf->mddev);
} elseif (bio != NULL && bio->bi_status) {
fail = true; if (!narrow_write_error(r10_bio, m))
md_error(conf->mddev, rdev);
rdev_dec_pending(rdev, conf->mddev);
}
bio = r10_bio->devs[m].repl_bio;
rdev = conf->mirrors[dev].replacement; if (rdev && bio == IO_MADE_GOOD) {
rdev_clear_badblocks(
rdev,
r10_bio->devs[m].addr,
r10_bio->sectors, 0);
rdev_dec_pending(rdev, conf->mddev);
}
} if (fail) {
spin_lock_irq(&conf->device_lock);
list_add(&r10_bio->retry_list, &conf->bio_end_io_list);
conf->nr_queued++;
spin_unlock_irq(&conf->device_lock); /* *Incasefreeze_array()iswaitingforcondition *nr_pending==nr_queued+extratobetrue.
*/
wake_up(&conf->wait_barrier);
md_wakeup_thread(conf->mddev->thread);
} else { if (test_bit(R10BIO_WriteError,
&r10_bio->state))
close_write(r10_bio);
raid_end_bio_io(r10_bio);
}
}
}
/* If we aborted, we need to abort the *synconthe'current'bitmapchucks(therecan *beseveralwhenrecoveringmultipledevices). *aswemayhavestartedsyncingitbutnotfinished. *Wecanfindthecurrentaddressin *mddev->curr_resync,butforrecovery, *weneedtoconvertthattoseveral *virtualaddresses.
*/ if (test_bit(MD_RECOVERY_RESHAPE, &mddev->recovery)) {
end_reshape(conf);
close_sync(conf); return0;
}
if (mddev->curr_resync < max_sector) { /* aborted */ if (test_bit(MD_RECOVERY_SYNC, &mddev->recovery))
mddev->bitmap_ops->end_sync(mddev,
mddev->curr_resync,
&sync_blocks); elsefor (i = 0; i < conf->geo.raid_disks; i++) {
sector_t sect =
raid10_find_virt(conf, mddev->curr_resync, i);
mddev->bitmap_ops->end_sync(mddev, sect,
&sync_blocks);
}
} else { /* completed sync */ if ((!mddev->bitmap || conf->fullsync)
&& conf->have_replacement
&& test_bit(MD_RECOVERY_SYNC, &mddev->recovery)) { /* Completed a full sync so the replacements *arenowfullyrecovered.
*/ for (i = 0; i < conf->geo.raid_disks; i++) { struct md_rdev *rdev =
conf->mirrors[i].replacement;
if (max_sector > mddev->resync_max)
max_sector = mddev->resync_max; /* Don't do IO beyond here */
/* make sure whole request will fit in a chunk - if chunks *aremeaningful
*/ if (conf->geo.near_copies < conf->geo.raid_disks &&
max_sector > (sector_nr | chunk_mask))
max_sector = (sector_nr | chunk_mask) + 1;
/* *Ifthereisnon-resyncactivitywaitingforaturn,thenletit *thoughbeforestartingonthisnewsyncrequest.
*/ if (conf->nr_waiting)
schedule_timeout_uninterruptible(1);
/* Again, very different code for resync and recovery. *Bothmustresultinanr10biowithalistofbiosthat *havebi_end_io,bi_sector,bi_bdevset, *andbi_privatesettother10bio. *Forrecovery,wemayactuallycreateseveralr10bios *with2biosineach,thatcorrespondtothebiosinthemainone. *Inthiscase,thesubordinater10bioslinkbackthrougha *borrowedmaster_biopointer,andthecounterinthemaster *includesareffromeachsubordinate.
*/ /* First, we decide what to do and set ->bi_end_io *Toend_sync_readifwewanttoread,and *end_sync_writeifwewillwanttowrite.
*/
max_sync = RESYNC_PAGES << (PAGE_SHIFT-9); if (!test_bit(MD_RECOVERY_SYNC, &mddev->recovery)) { /* recovery... the complicated one */ int j;
r10_bio = NULL;
for (i = 0 ; i < conf->geo.raid_disks; i++) { bool still_degraded; struct r10bio *rb2;
sector_t sect; bool must_sync; int any_working; struct raid10_info *mirror = &conf->mirrors[i]; struct md_rdev *mrdev, *mreplace;
any_working = 0; for (j=0; j<conf->copies;j++) { int k; int d = r10_bio->devs[j].devnum;
sector_t from_addr, to_addr; struct md_rdev *rdev = conf->mirrors[d].rdev;
sector_t sector, first_bad;
sector_t bad_sectors; if (!rdev ||
!test_bit(In_sync, &rdev->flags)) continue; /* This is where we read from */
any_working = 1;
sector = r10_bio->devs[j].addr;
if (is_badblock(rdev, sector, max_sync,
&first_bad, &bad_sectors)) { if (first_bad > sector)
max_sync = first_bad - sector; else {
bad_sectors -= (sector
- first_bad); if (max_sync > bad_sectors)
max_sync = bad_sectors; continue;
}
}
bio = r10_bio->devs[0].bio;
bio->bi_next = biolist;
biolist = bio;
bio->bi_end_io = end_sync_read;
bio->bi_opf = REQ_OP_READ; if (test_bit(FailFast, &rdev->flags))
bio->bi_opf |= MD_FAILFAST;
from_addr = r10_bio->devs[j].addr;
bio->bi_iter.bi_sector = from_addr +
rdev->data_offset;
bio_set_dev(bio, rdev->bdev);
atomic_inc(&rdev->nr_pending); /* and we write to 'i' (if not in_sync) */
/* and maybe write to replacement */
bio = r10_bio->devs[1].repl_bio; if (bio)
bio->bi_end_io = NULL; /* Note: if replace is not NULL, then bio *cannotbeNULLasr10buf_pool_allocwill *haveallocatedit.
*/ if (!mreplace) break;
bio->bi_next = biolist;
biolist = bio;
bio->bi_end_io = end_sync_write;
bio->bi_opf = REQ_OP_WRITE;
bio->bi_iter.bi_sector = to_addr +
mreplace->data_offset;
bio_set_dev(bio, mreplace->bdev);
atomic_inc(&r10_bio->remaining); break;
} if (j == conf->copies) { /* Cannot recover, so abort the recovery or
* record a bad block */ if (any_working) { /* problem is that there are bad blocks *onotherdevice(s)
*/ int k; for (k = 0; k < conf->copies; k++) if (r10_bio->devs[k].devnum == i) break; if (mrdev && !test_bit(In_sync,
&mrdev->flags)
&& !rdev_set_badblocks(
mrdev,
r10_bio->devs[k].addr,
max_sync, 0))
any_working = 0; if (mreplace &&
!rdev_set_badblocks(
mreplace,
r10_bio->devs[k].addr,
max_sync, 0))
any_working = 0;
} if (!any_working) { if (!test_and_set_bit(MD_RECOVERY_INTR,
&mddev->recovery))
pr_warn("md/raid10:%s: insufficient working devices for recovery.\n",
mdname(mddev));
mirror->recovery_disabled
= mddev->recovery_disabled;
} else {
error_disk = i;
}
put_buf(r10_bio); if (rb2)
atomic_dec(&rb2->remaining);
r10_bio = rb2; if (mrdev)
rdev_dec_pending(mrdev, mddev); if (mreplace)
rdev_dec_pending(mreplace, mddev); break;
} if (mrdev)
rdev_dec_pending(mrdev, mddev); if (mreplace)
rdev_dec_pending(mreplace, mddev); if (r10_bio->devs[0].bio->bi_opf & MD_FAILFAST) { /* Only want this if there is elsewhere to *readfrom.'j'iscurrentlythefirst *readablecopy.
*/ int targets = 1; for (; j < conf->copies; j++) { int d = r10_bio->devs[j].devnum; if (conf->mirrors[d].rdev &&
test_bit(In_sync,
&conf->mirrors[d].rdev->flags))
targets++;
} if (targets == 1)
r10_bio->devs[0].bio->bi_opf
&= ~MD_FAILFAST;
}
} if (biolist == NULL) { while (r10_bio) { struct r10bio *rb2 = r10_bio;
r10_bio = (struct r10bio*) rb2->master_bio;
rb2->master_bio = NULL;
put_buf(rb2);
} goto giveup;
}
} else { /* resync. Schedule a read for every block at this virt offset */ int count = 0;
if (sectors_skipped) /* pretend they weren't skipped, it makes *noimportantdifferenceinthiscase
*/
md_done_sync(mddev, sectors_skipped, 1);
return sectors_skipped + nr_sectors;
giveup: /* There is nowhere to write, so all non-sync *drivesmustbefailedorinresync,alldrives *haveabadblock,sotrythenextchunk...
*/ if (sector_nr + max_sync < max_sector)
max_sector = sector_nr + max_sync;
staticvoid calc_sectors(struct r10conf *conf, sector_t size)
{ /* Calculate the number of sectors-per-device that will *actuallybeused,andsetconf->dev_sectorsand *conf->stride
*/
size = size >> conf->geo.chunk_shift;
sector_div(size, conf->geo.far_copies);
size = size * conf->geo.raid_disks;
sector_div(size, conf->geo.near_copies); /* 'size' is now the number of chunks in the array */ /* calculate "used chunks per device" */
size = size * conf->copies;
/* We need to round up when dividing by raid_disks to *getthestridesize.
*/
size = DIV_ROUND_UP_SECTOR_T(size, conf->geo.raid_disks);
if (mddev_is_clustered(conf->mddev)) { int fc, fo;
fc = (mddev->layout >> 8) & 255;
fo = mddev->layout & (1<<16); if (fc > 1 || fo > 0) {
pr_err("only near layout is supported by clustered" " raid10\n"); goto out_free_conf;
}
}
rdev_for_each(rdev, mddev) { longlong diff;
disk_idx = rdev->raid_disk; if (disk_idx < 0) continue; if (disk_idx >= conf->geo.raid_disks &&
disk_idx >= conf->prev.raid_disks) continue;
disk = conf->mirrors + disk_idx;
if (test_bit(Replacement, &rdev->flags)) { if (disk->replacement) goto out_free_conf;
disk->replacement = rdev;
} else { if (disk->rdev) goto out_free_conf;
disk->rdev = rdev;
}
diff = (rdev->new_data_offset - rdev->data_offset); if (!mddev->reshape_backwards)
diff = -diff; if (diff < 0)
diff = 0; if (first || diff < min_offset_diff)
min_offset_diff = diff;
disk->head_position = 0;
first = 0;
}
if (!mddev_is_dm(conf->mddev)) { int err = raid10_set_queue_limits(mddev);
if (err) {
ret = err; goto out_free_conf;
}
}
/* need to check that every block has at least one working mirror */ if (!enough(conf, -1)) {
pr_err("md/raid10:%s: not enough operational mirrors.\n",
mdname(mddev)); goto out_free_conf;
}
if (conf->reshape_progress != MaxSector) { /* must ensure that shape change is supported */ if (conf->geo.far_copies != 1 &&
conf->geo.far_offset == 0) goto out_free_conf; if (conf->prev.far_copies != 1 &&
conf->prev.far_offset == 0) goto out_free_conf;
}
mddev->degraded = 0; for (i = 0;
i < conf->geo.raid_disks
|| i < conf->prev.raid_disks;
i++) {
disk = conf->mirrors + i;
if (!disk->rdev && disk->replacement) { /* The replacement is all we have - use it */
disk->rdev = disk->replacement;
disk->replacement = NULL;
clear_bit(Replacement, &disk->rdev->flags);
}
if (!disk->rdev ||
!test_bit(In_sync, &disk->rdev->flags)) {
disk->head_position = 0;
mddev->degraded++; if (disk->rdev &&
disk->rdev->saved_raid_disk < 0)
conf->fullsync = 1;
}
if (max(before_length, after_length) > min_offset_diff) { /* This cannot work */
pr_warn("md/raid10: offset difference not enough to continue reshape\n"); goto out_free_conf;
}
conf->offset_diff = min_offset_diff;
/* raid10 can take over: *raid0-providingithasonlytwodrives
*/ if (mddev->level == 0) { /* for raid0 takeover only one zone is supported */
raid0_conf = mddev->private; if (raid0_conf->nr_strip_zones > 1) {
pr_warn("md/raid10:%s: cannot takeover raid 0 with more than one zone.\n",
mdname(mddev)); return ERR_PTR(-EINVAL);
} return raid10_takeover_raid0(mddev,
raid0_conf->strip_zone->zone_end,
raid0_conf->strip_zone->nb_dev);
} return ERR_PTR(-EINVAL);
}
staticint raid10_check_reshape(struct mddev *mddev)
{ /* Called when there is a request to change *-layout(to->new_layout) *-chunksize(to->new_chunk_sectors) *-raid_disks(bydelta_disks) *orwhentryingtorestartareshapethatwasongoing. * *Weneedtovalidatetherequestandpossiblyallocate *spaceifthatmightbeanissuelater. * *Currentlywerejectanyreshapeofa'far'modearray, *allowchunksizetochangeifnewisgenerallyacceptable, *allowraid_diskstoincrease,andallow *aswitchbetween'near'modeand'offset'mode.
*/ struct r10conf *conf = mddev->private; struct geom geo;
if (conf->geo.far_copies != 1 && !conf->geo.far_offset) return -EINVAL;
if (setup_geo(&geo, mddev, geo_start) != conf->copies) /* mustn't change number of copies */ return -EINVAL; if (geo.far_copies > 1 && !geo.far_offset) /* Cannot switch to 'far' mode */ return -EINVAL;
if (mddev->array_sectors & geo.chunk_mask) /* not factor of array size */ return -EINVAL;
if (!enough(conf, -1)) return -EINVAL;
kfree(conf->mirrors_new);
conf->mirrors_new = NULL; if (mddev->delta_disks > 0) { /* allocate new 'mirrors' list */
conf->mirrors_new =
kcalloc(mddev->raid_disks + mddev->delta_disks, sizeof(struct raid10_info),
GFP_KERNEL); if (!conf->mirrors_new) return -ENOMEM;
} return0;
}
degraded = 0; /* 'prev' section first */ for (i = 0; i < conf->prev.raid_disks; i++) { struct md_rdev *rdev = conf->mirrors[i].rdev;
if (!rdev || test_bit(Faulty, &rdev->flags))
degraded++; elseif (!test_bit(In_sync, &rdev->flags)) /* When we can reduce the number of devices in *anarray,thismightnotcontributeto *'degraded'.Itdoesnow.
*/
degraded++;
} if (conf->geo.raid_disks == conf->prev.raid_disks) return degraded;
degraded2 = 0; for (i = 0; i < conf->geo.raid_disks; i++) { struct md_rdev *rdev = conf->mirrors[i].rdev;
if (!rdev || test_bit(Faulty, &rdev->flags))
degraded2++; elseif (!test_bit(In_sync, &rdev->flags)) { /* If reshape is increasing the number of devices, *thissectionhasalreadybeenrecovered,so *itdoesn'tcontributetodegraded. *elseitdoes.
*/ if (conf->geo.raid_disks <= conf->prev.raid_disks)
degraded2++;
}
} if (degraded2 > degraded) return degraded2; return degraded;
}
staticint raid10_start_reshape(struct mddev *mddev)
{ /* A 'reshape' has been requested. This commits *thevarious'new'fieldsandsetsMD_RECOVER_RESHAPE *Thisalsochecksifthereareenoughsparesandaddsthem *tothearray. *Wecurrentlyrequireenoughsparestomakethefinal *arraynon-degraded.Wealsorequirethatthedifference *betweenoldandnewdata_offset-oneachdevice-is *enoughthatweneverriskover-writing.
*/
unsignedlong before_length, after_length;
sector_t min_offset_diff = 0; int first = 1; struct geom new; struct r10conf *conf = mddev->private; struct md_rdev *rdev; int spares = 0; int ret;
if (test_bit(MD_RECOVERY_RUNNING, &mddev->recovery)) return -EBUSY;
if (setup_geo(&new, mddev, geo_start) != conf->copies) return -EINVAL;
ret = mddev->bitmap_ops->resize(mddev, newsize, 0, false); if (ret) goto abort;
ret = mddev->cluster_ops->resize_bitmaps(mddev, newsize, oldsize); if (ret) {
mddev->bitmap_ops->resize(mddev, oldsize, 0, false); goto abort;
}
}
out: if (mddev->delta_disks > 0) {
rdev_for_each(rdev, mddev) if (rdev->raid_disk < 0 &&
!test_bit(Faulty, &rdev->flags)) { if (raid10_add_disk(mddev, rdev) == 0) { if (rdev->raid_disk >=
conf->prev.raid_disks)
set_bit(In_sync, &rdev->flags); else
rdev->recovery_offset = 0;
/* Failure here is OK */
sysfs_link_rdev(mddev, rdev);
}
} elseif (rdev->raid_disk >= conf->prev.raid_disks
&& !test_bit(Faulty, &rdev->flags)) { /* This is a spare that was manually added */
set_bit(In_sync, &rdev->flags);
}
} /* When a reshape changes the number of devices, *->degradedismeasuredagainstthelargerofthe *preandpostnumbers.
*/
spin_lock_irq(&conf->device_lock);
mddev->degraded = calc_degraded(conf);
spin_unlock_irq(&conf->device_lock);
mddev->raid_disks = conf->geo.raid_disks;
mddev->reshape_position = conf->reshape_progress;
set_bit(MD_SB_CHANGE_DEVS, &mddev->sb_flags);
/* Calculate the last device-address that could contain *anyblockfromthechunkthatincludesthearray-address's' *andreportthenextaddress. *i.e.theaddressreturnedwillbechunk-alignedandafter *anydatathatisinthechunkcontaining's'.
*/ static sector_t last_dev_address(sector_t s, struct geom *geo)
{
s = (s | geo->chunk_mask) + 1;
s >>= geo->chunk_shift;
s *= geo->near_copies;
s = DIV_ROUND_UP_SECTOR_T(s, geo->raid_disks);
s *= geo->far_copies;
s <<= geo->chunk_shift; return s;
}
/* Calculate the first device-address that could contain *anyblockfromthechunkthatincludesthearray-address's'. *Thistoowillbethestartofachunk
*/ static sector_t first_dev_address(sector_t s, struct geom *geo)
{
s >>= geo->chunk_shift;
s *= geo->near_copies;
sector_div(s, geo->raid_disks);
s *= geo->far_copies;
s <<= geo->chunk_shift; return s;
}
static sector_t reshape_request(struct mddev *mddev, sector_t sector_nr, int *skipped)
{ /* We simply copy at most one chunk (smallest of old and new) *atatime,possiblylessifthatexceedsRESYNC_PAGES, *orwehitabadblockorsomething. *ThismightmeanwepausefornormalIOinthemiddleof *achunk,butthatisnotaproblemasmddev->reshape_position *canrecordanylocation. * *Ifwewillwanttowritetoalocationthatisn't *yetrecordedas'safe'(i.e.inmetadataondisk)then *weneedtoflushallreshaperequestsandupdatethemetadata. * *Whenreshapingforwards(e.g.tomoredevices),weinterpret *'safe'astheearliestblockwhichmightnothavebeencopied *downyet.Wedividethisbypreviousstripesizeandmultiply *bypreviousstripelengthtogetlowestdeviceoffsetthatwe *cannotwritetoyet. *Weinterpret'sector_nr'asanaddressthatwewanttowriteto. *Fromthisweuselast_device_address()tofindwherewemight *writeto,andfirst_device_addressonthe'safe'position. *Ifthis'next'writepositionisafterthe'safe'position, *wemustupdatethemetadatatoincreasethe'safe'position. * *Whenreshapingbackwards,weroundintheoppositedirection *andperformthereversetest:nextwritepositionmustnotbe *lessthancurrentsafeposition. * *Inallthistheminimumdifferenceindataoffsets *(conf->offset_diff-alwayspositive)allowsabitofslack, *sonextcanbeafter'safe',butnotbymorethanoffset_diff * *WeneedtoprepareallthebiosherebeforewestartanyIO *toensurethesizewechooseisacceptabletoalldevices. *Themeansoneforeachcopyforwrite-outandanextraonefor *read-in. *Westoretheread-inbioin->master_bioandtheothersin *->devs[x].bioand->devs[x].repl_bio.
*/ struct r10conf *conf = mddev->private; struct r10bio *r10_bio;
sector_t next, safe, last; int max_sectors; int nr_sectors; int s; struct md_rdev *rdev; int need_flush = 0; struct bio *blist; struct bio *bio, *read_bio; int sectors_done = 0; struct page **pages;
if (sector_nr == 0) { /* If restarting in the middle, skip the initial sectors */ if (mddev->reshape_backwards &&
conf->reshape_progress < raid10_size(mddev, 0, 0)) {
sector_nr = (raid10_size(mddev, 0, 0)
- conf->reshape_progress);
} elseif (!mddev->reshape_backwards &&
conf->reshape_progress > 0)
sector_nr = conf->reshape_progress; if (sector_nr) {
mddev->curr_resync_completed = sector_nr;
sysfs_notify_dirent_safe(mddev->sysfs_completed);
*skipped = 1; return sector_nr;
}
}
/* We don't use sector_nr to track where we are up to *asthatdoesn'tworkwellfor->reshape_backwards. *Sojustuse->reshape_progress.
*/ if (mddev->reshape_backwards) { /* 'next' is the earliest device address that we might *writetoforthischunkinthenewlayout
*/
next = first_dev_address(conf->reshape_progress - 1,
&conf->geo);
/* 'safe' is the last device address that we might read from *intheoldlayoutafterarestart
*/
safe = last_dev_address(conf->reshape_safe - 1,
&conf->prev);
if (next + conf->offset_diff < safe)
need_flush = 1;
last = conf->reshape_progress - 1;
sector_nr = last & ~(sector_t)(conf->geo.chunk_mask
& conf->prev.chunk_mask); if (sector_nr + RESYNC_SECTORS < last)
sector_nr = last + 1 - RESYNC_SECTORS;
} else { /* 'next' is after the last device address that we *mightwritetoforthischunkinthenewlayout
*/
next = last_dev_address(conf->reshape_progress, &conf->geo);
/* 'safe' is the earliest device address that we might *readfromintheoldlayoutafterarestart
*/
safe = first_dev_address(conf->reshape_safe, &conf->prev);
/* Need to update metadata if 'next' might be beyond 'safe' *asthatwouldpossiblycorruptdata
*/ if (next > safe + conf->offset_diff)
need_flush = 1;
sector_nr = conf->reshape_progress;
last = sector_nr | (conf->geo.chunk_mask
& conf->prev.chunk_mask);
if (sector_nr + RESYNC_SECTORS <= last)
last = sector_nr + RESYNC_SECTORS - 1;
}
if (need_flush ||
time_after(jiffies, conf->reshape_checkpoint + 10*HZ)) { /* Need to update reshape_position in metadata */
wait_barrier(conf, false);
mddev->reshape_position = conf->reshape_progress; if (mddev->reshape_backwards)
mddev->curr_resync_completed = raid10_size(mddev, 0, 0)
- conf->reshape_progress; else
mddev->curr_resync_completed = conf->reshape_progress;
conf->reshape_checkpoint = jiffies;
set_bit(MD_SB_CHANGE_DEVS, &mddev->sb_flags);
md_wakeup_thread(mddev->thread);
wait_event(mddev->sb_wait, mddev->sb_flags == 0 ||
test_bit(MD_RECOVERY_INTR, &mddev->recovery)); if (test_bit(MD_RECOVERY_INTR, &mddev->recovery)) {
allow_barrier(conf); return sectors_done;
}
conf->reshape_safe = mddev->reshape_position;
allow_barrier(conf);
}
raise_barrier(conf, 0);
read_more: /* Now schedule reads for blocks from sector_nr to last */
r10_bio = raid10_alloc_init_r10buf(conf);
r10_bio->state = 0;
raise_barrier(conf, 1);
atomic_set(&r10_bio->remaining, 0);
r10_bio->mddev = mddev;
r10_bio->sector = sector_nr;
set_bit(R10BIO_IsReshape, &r10_bio->state);
r10_bio->sectors = last - sector_nr + 1;
rdev = read_balance(conf, r10_bio, &max_sectors);
BUG_ON(!test_bit(R10BIO_Previous, &r10_bio->state));
if (!rdev) { /* Cannot read from here, so need to record bad blocks *onallthetargetdevices.
*/ // FIXME
mempool_free(r10_bio, &conf->r10buf_pool);
set_bit(MD_RECOVERY_INTR, &mddev->recovery); return sectors_done;
}
/* Now find the locations in the new layout */
__raid10_find_phys(&conf->geo, r10_bio);
blist = read_bio;
read_bio->bi_next = NULL;
for (s = 0; s < conf->copies*2; s++) { struct bio *b; int d = r10_bio->devs[s/2].devnum; struct md_rdev *rdev2; if (s&1) {
rdev2 = conf->mirrors[d].replacement;
b = r10_bio->devs[s/2].repl_bio;
} else {
rdev2 = conf->mirrors[d].rdev;
b = r10_bio->devs[s/2].bio;
} if (!rdev2 || test_bit(Faulty, &rdev2->flags)) continue;
/* Now add as many pages as possible to all of these bios. */
nr_sectors = 0;
pages = get_resync_pages(r10_bio->devs[0].bio)->pages; for (s = 0 ; s < max_sectors; s += PAGE_SIZE >> 9) { struct page *page = pages[s / (PAGE_SIZE >> 9)]; int len = (max_sectors - s) << 9; if (len > PAGE_SIZE)
len = PAGE_SIZE; for (bio = blist; bio ; bio = bio->bi_next) { if (WARN_ON(!bio_add_page(bio, page, len, 0))) {
bio->bi_status = BLK_STS_RESOURCE;
bio_endio(bio); return sectors_done;
}
}
sector_nr += len >> 9;
nr_sectors += len >> 9;
}
r10_bio->sectors = nr_sectors;
/* Now submit the read */
atomic_inc(&r10_bio->remaining);
read_bio->bi_next = NULL;
submit_bio_noacct(read_bio);
sectors_done += nr_sectors; if (sector_nr <= last) goto read_more;
lower_barrier(conf);
/* Now that we have done the whole section we can *updatereshape_progress
*/ if (mddev->reshape_backwards)
conf->reshape_progress -= sectors_done; else
conf->reshape_progress += sectors_done;
return sectors_done;
}
staticvoid end_reshape_request(struct r10bio *r10_bio); staticint handle_reshape_read_error(struct mddev *mddev, struct r10bio *r10_bio); staticvoid reshape_request_write(struct mddev *mddev, struct r10bio *r10_bio)
{ /* Reshape read completed. Hopefully we have a block *towriteout. *Ifwegotareaderrorthenwedosync1-pagereadsfrom *elsewhereuntilwefindthedata-orgiveup.
*/ struct r10conf *conf = mddev->private; int s;
if (!test_bit(R10BIO_Uptodate, &r10_bio->state)) if (handle_reshape_read_error(mddev, r10_bio) < 0) { /* Reshape has been aborted */
md_done_sync(mddev, r10_bio->sectors, 0); return;
}
/* We definitely have the data in the pages, schedule the *writes.
*/
atomic_set(&r10_bio->remaining, 1); for (s = 0; s < conf->copies*2; s++) { struct bio *b; int d = r10_bio->devs[s/2].devnum; struct md_rdev *rdev; if (s&1) {
rdev = conf->mirrors[d].replacement;
b = r10_bio->devs[s/2].repl_bio;
} else {
rdev = conf->mirrors[d].rdev;
b = r10_bio->devs[s/2].bio;
} if (!rdev || test_bit(Faulty, &rdev->flags)) continue;
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.301Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-09-29)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.