staticbool devices_handle_discard_safely = false;
module_param(devices_handle_discard_safely, bool, 0644);
MODULE_PARM_DESC(devices_handle_discard_safely, "Set to Y if all devices in each array reliably return zeroes on reads from discarded regions"); staticstruct workqueue_struct *raid5_wq;
staticvoid raid5_quiesce(struct mddev *mddev, int quiesce);
staticinlinevoid lock_all_device_hash_locks_irq(struct r5conf *conf)
__acquires(&conf->device_lock)
{ int i;
spin_lock_irq(conf->hash_locks); for (i = 1; i < NR_STRIPE_HASH_LOCKS; i++)
spin_lock_nest_lock(conf->hash_locks + i, conf->hash_locks);
spin_lock(&conf->device_lock);
}
staticinlinevoid unlock_all_device_hash_locks_irq(struct r5conf *conf)
__releases(&conf->device_lock)
{ int i;
spin_unlock(&conf->device_lock); for (i = NR_STRIPE_HASH_LOCKS - 1; i; i--)
spin_unlock(conf->hash_locks + i);
spin_unlock_irq(conf->hash_locks);
}
/* Find first data disk in a raid6 stripe */ staticinlineint raid6_d0(struct stripe_head *sh)
{ if (sh->ddf_layout) /* ddf always start from first device */ return0; /* md starts just after Q block */ if (sh->qd_idx == sh->disks - 1) return0; else return sh->qd_idx + 1;
} staticinlineint raid6_next_disk(int disk, int raid_disks)
{
disk++; return (disk < raid_disks) ? disk : 0;
}
/* When walking through the disks in a raid5, starting at raid6_d0, *Weneedtomapeachdisktoa'slot',wherethedatadisksareslot *0..raid_disks-3,theparitydiskisraid_disks-2andtheQdisk *israid_disks-1.Thishelpdoesthatmapping.
*/ staticint raid6_idx_to_slot(int idx, struct stripe_head *sh, int *count, int syndrome_disks)
{ int slot = *count;
if (sh->ddf_layout)
(*count)++; if (idx == sh->pd_idx) return syndrome_disks; if (idx == sh->qd_idx) return syndrome_disks + 1; if (!sh->ddf_layout)
(*count)++; return slot;
}
staticvoid raid5_wakeup_stripe_thread(struct stripe_head *sh)
__must_hold(&sh->raid_conf->device_lock)
{ struct r5conf *conf = sh->raid_conf; struct r5worker_group *group; int thread_cnt; int i, cpu = sh->cpu;
if (!cpu_online(cpu)) {
cpu = cpumask_any(cpu_online_mask);
sh->cpu = cpu;
}
if (list_empty(&sh->lru)) { struct r5worker_group *group;
group = conf->worker_groups + cpu_to_group(cpu); if (stripe_is_lowprio(sh))
list_add_tail(&sh->lru, &group->loprio_list); else
list_add_tail(&sh->lru, &group->handle_list);
group->stripes_cnt++;
sh->group = group;
}
if (conf->worker_cnt_per_group == 0) {
md_wakeup_thread(conf->mddev->thread); return;
}
group = conf->worker_groups + cpu_to_group(sh->cpu);
group->workers[0].working = true; /* at least one worker should run to avoid race */
queue_work_on(sh->cpu, raid5_wq, &group->workers[0].work);
thread_cnt = group->stripes_cnt / MAX_STRIPE_BATCH - 1; /* wakeup more workers */ for (i = 1; i < conf->worker_cnt_per_group && thread_cnt > 0; i++) { if (group->workers[i].working == false) {
group->workers[i].working = true;
queue_work_on(sh->cpu, raid5_wq,
&group->workers[i].work);
thread_cnt--;
}
}
}
staticvoid do_release_stripe(struct r5conf *conf, struct stripe_head *sh, struct list_head *temp_inactive_list)
__must_hold(&conf->device_lock)
{ int i; int injournal = 0; /* number of date pages with R5_InJournal */
if (do_wakeup) {
wake_up(&conf->wait_for_stripe); if (atomic_read(&conf->active_stripes) == 0)
wake_up(&conf->wait_for_quiescent); if (conf->retry_read_aligned)
md_wakeup_thread(conf->mddev->thread);
}
}
/* Avoid release_list until the last reference.
*/ if (atomic_add_unless(&sh->count, -1, 1)) return;
if (unlikely(!conf->mddev->thread) ||
test_and_set_bit(STRIPE_ON_RELEASE_LIST, &sh->state)) goto slow_path;
wakeup = llist_add(&sh->release_list, &conf->released_stripes); if (wakeup)
md_wakeup_thread(conf->mddev->thread); return;
slow_path: /* we are ok here if STRIPE_ON_RELEASE_LIST is set or not */ if (atomic_dec_and_lock_irqsave(&sh->count, &conf->device_lock, flags)) {
INIT_LIST_HEAD(&list);
hash = sh->hash_lock_index;
do_release_stripe(conf, sh, &list);
spin_unlock_irqrestore(&conf->device_lock, flags);
release_inactive_stripe_list(conf, &list, hash);
}
}
/* find an idle stripe, make sure it is unhashed, and return it. */ staticstruct stripe_head *get_free_stripe(struct r5conf *conf, int hash)
{ struct stripe_head *sh = NULL; struct list_head *first;
if (list_empty(conf->inactive_list + hash)) goto out;
first = (conf->inactive_list + hash)->next;
sh = list_entry(first, struct stripe_head, lru);
list_del_init(first);
remove_hash(sh);
atomic_inc(&conf->active_stripes);
BUG_ON(hash != sh->hash_lock_index); if (list_empty(conf->inactive_list + hash))
atomic_inc(&conf->empty_inactive_list_nr);
out: return sh;
}
/* Only freshly new full stripe normal write stripe can be added to a batch list */ staticbool stripe_can_batch(struct stripe_head *sh)
{ struct r5conf *conf = sh->raid_conf;
/* we only do back search */ staticvoid stripe_add_to_batch_list(struct r5conf *conf, struct stripe_head *sh, struct stripe_head *last_sh)
{ struct stripe_head *head;
sector_t head_sector, tmp_sec; int hash; int dd_idx;
/* Don't cross chunks, so stripe pd_idx/qd_idx is the same */
tmp_sec = sh->sector; if (!sector_div(tmp_sec, conf->chunk_sectors)) return;
head_sector = sh->sector - RAID5_STRIPE_SECTORS(conf);
if (last_sh && head_sector == last_sh->sector) {
head = last_sh;
atomic_inc(&head->count);
} else {
hash = stripe_hash_locks_hash(conf, head_sector);
spin_lock_irq(conf->hash_locks + hash);
head = find_get_stripe(conf, head_sector, conf->generation,
hash);
spin_unlock_irq(conf->hash_locks + hash); if (!head) return; if (!stripe_can_batch(head)) goto out;
}
lock_two_stripes(head, sh); /* clear_batch_ready clear the flag */ if (!stripe_can_batch(head) || !stripe_can_batch(sh)) goto unlock_out;
if (test_and_clear_bit(STRIPE_PREREAD_ACTIVE, &sh->state)) if (atomic_dec_return(&conf->preread_active_stripes)
< IO_THRESHOLD)
md_wakeup_thread(conf->mddev->thread);
if (test_and_clear_bit(STRIPE_BIT_DELAY, &sh->state)) { int seq = sh->bm_seq; if (test_bit(STRIPE_BIT_DELAY, &sh->batch_head->state) &&
sh->batch_head->bm_seq > seq)
seq = sh->batch_head->bm_seq;
set_bit(STRIPE_BIT_DELAY, &sh->batch_head->state);
sh->batch_head->bm_seq = seq;
}
/* Determine if 'data_offset' or 'new_data_offset' should be used *inthisstripe_head.
*/ staticint use_new_offset(struct r5conf *conf, struct stripe_head *sh)
{
sector_t progress = conf->reshape_progress; /* Need a memory barrier to make sure we see the value *ofconf->generation,or->data_offsetthatwassetbefore *reshape_progresswasupdated.
*/
smp_rmb(); if (progress == MaxSector) return0; if (sh->generation == conf->generation - 1) return0; /* We are in a reshape, and this is a new-generation stripe, *sousenew_data_offset.
*/ return1;
}
staticvoid dispatch_bio_list(struct bio_list *tmp)
{ struct bio *bio;
while ((bio = bio_list_pop(tmp)))
submit_bio_noacct(bio);
}
/* temporarily move the head */ if (conf->next_pending_data)
list_move_tail(&conf->pending_list,
&conf->next_pending_data->sibling);
while (!list_empty(&conf->pending_list)) {
data = list_first_entry(&conf->pending_list, struct r5pending_data, sibling); if (&data->sibling == first)
first = data->sibling.next;
next = data->sibling.next;
for (i = disks; i--; ) { enum req_op op;
blk_opf_t op_flags = 0; int replace_only = 0; struct bio *bi, *rbi; struct md_rdev *rdev, *rrdev = NULL;
sh = head_sh; if (test_and_clear_bit(R5_Wantwrite, &sh->dev[i].flags)) {
op = REQ_OP_WRITE; if (test_and_clear_bit(R5_WantFUA, &sh->dev[i].flags))
op_flags = REQ_FUA; if (test_bit(R5_Discard, &sh->dev[i].flags))
op = REQ_OP_DISCARD;
} elseif (test_and_clear_bit(R5_Wantread, &sh->dev[i].flags))
op = REQ_OP_READ; elseif (test_and_clear_bit(R5_WantReplace,
&sh->dev[i].flags)) {
op = REQ_OP_WRITE;
replace_only = 1;
} else continue; if (test_and_clear_bit(R5_SyncIO, &sh->dev[i].flags))
op_flags |= REQ_SYNC;
again:
dev = &sh->dev[i];
bi = &dev->req;
rbi = &dev->rreq; /* For writing to replacement */
rdev = conf->disks[i].rdev;
rrdev = conf->disks[i].replacement; if (op_is_write(op)) { if (replace_only)
rdev = NULL; if (rdev == rrdev) /* We raced and saw duplicates */
rrdev = NULL;
} else { if (test_bit(R5_ReadRepl, &head_sh->dev[i].flags) && rrdev)
rdev = rrdev;
rrdev = NULL;
}
if (rdev && test_bit(Faulty, &rdev->flags))
rdev = NULL; if (rdev)
atomic_inc(&rdev->nr_pending); if (rrdev && test_bit(Faulty, &rrdev->flags))
rrdev = NULL; if (rrdev)
atomic_inc(&rrdev->nr_pending);
/* We have already checked bad blocks for reads. Now *needtocheckforwrites.Weneveracceptwriteerrors *onthereplacement,sowedon'ttocheckrrdev.
*/ while (op_is_write(op) && rdev &&
test_bit(WriteErrorSeen, &rdev->flags)) { int bad = rdev_has_badblock(rdev, sh->sector,
RAID5_STRIPE_SECTORS(conf)); if (!bad) break;
if (bad < 0) {
set_bit(BlockedBadBlocks, &rdev->flags); if (!conf->mddev->external &&
conf->mddev->sb_flags) { /* It is very unlikely, but we might *stillneedtowriteoutthe *badblocklog-bettergiveit
* a chance*/
md_check_recovery(conf->mddev);
} /* *Becausemd_wait_for_blocked_rdev *willdecnr_pending,wemust *incrementitfirst.
*/
atomic_inc(&rdev->nr_pending);
md_wait_for_blocked_rdev(rdev, conf->mddev);
} else { /* Acknowledged bad block - skip the write */
rdev_dec_pending(rdev, conf->mddev);
rdev = NULL;
}
}
if (rdev) {
set_bit(STRIPE_IO_STARTED, &sh->state);
/* clear completed biofills */ for (i = sh->disks; i--; ) { struct r5dev *dev = &sh->dev[i];
/* acknowledge completion of a biofill operation */ /* and check if we need to reply to a read request, *newR5_Wantfillrequestsareheldoffuntil *!STRIPE_BIOFILL_RUN
*/ if (test_and_clear_bit(R5_Wantfill, &dev->flags)) { struct bio *rbi, *rbi2;
/* return a pointer to the address conversion region of the scribble buffer */ staticstruct page **to_addr_page(struct raid5_percpu *percpu, int i)
{ return percpu->scribble + i * percpu->scribble_obj_size;
}
/* return a pointer to the address conversion region of the scribble buffer */ static addr_conv_t *to_addr_conv(struct stripe_head *sh, struct raid5_percpu *percpu, int i)
{ return (void *) (to_addr_page(percpu, i) + sh->disks + 2);
}
/* we need to open-code set_syndrome_sources to handle the *slotnumberconversionfor'faila'and'failb'
*/ for (i = 0; i < disks ; i++) {
offs[i] = 0;
blocks[i] = NULL;
}
count = 0;
i = d0_idx; do { int slot = raid6_idx_to_slot(i, sh, &count, syndrome_disks);
for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i]; /* Only process blocks that are known to be uptodate */ if (test_bit(R5_InJournal, &dev->flags)) { /* *Forthiscase,PAGE_SIZEmustbeequalto4KBand *pageoffsetiszero.
*/
off_srcs[count] = dev->offset;
xor_srcs[count++] = dev->orig_page;
} elseif (test_bit(R5_Wantdrain, &dev->flags)) {
off_srcs[count] = dev->offset;
xor_srcs[count++] = dev->page;
}
}
for (i = disks; i--; ) {
fua |= test_bit(R5_WantFUA, &sh->dev[i].flags);
sync |= test_bit(R5_SyncIO, &sh->dev[i].flags);
discard |= test_bit(R5_Discard, &sh->dev[i].flags);
}
for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i];
if (dev->written || i == pd_idx || i == qd_idx) { if (!discard && !test_bit(R5_SkipCopy, &dev->flags)) {
set_bit(R5_UPTODATE, &dev->flags); if (test_bit(STRIPE_EXPAND_READY, &sh->state))
set_bit(R5_Expanded, &dev->flags);
} if (fua)
set_bit(R5_WantFUA, &dev->flags); if (sync)
set_bit(R5_SyncIO, &dev->flags);
}
}
for (i = 0; i < sh->disks; i++) { if (pd_idx == i) continue; if (!test_bit(R5_Discard, &sh->dev[i].flags)) break;
} if (i >= sh->disks) {
atomic_inc(&sh->count);
set_bit(R5_Discard, &sh->dev[pd_idx].flags);
ops_complete_reconstruct(sh); return;
}
again:
count = 0;
xor_srcs = to_addr_page(percpu, j);
off_srcs = to_addr_offs(sh, percpu); /* check if prexor is active which means only process blocks *thatarepartofaread-modify-write(written)
*/ if (head_sh->reconstruct_state == reconstruct_state_prexor_drain_run) {
prexor = 1;
off_dest = off_srcs[count] = sh->dev[pd_idx].offset;
xor_dest = xor_srcs[count++] = sh->dev[pd_idx].page; for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i]; if (head_sh->dev[i].written ||
test_bit(R5_InJournal, &head_sh->dev[i].flags)) {
off_srcs[count] = dev->offset;
xor_srcs[count++] = dev->page;
}
}
} else {
xor_dest = sh->dev[pd_idx].page;
off_dest = sh->dev[pd_idx].offset; for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i]; if (i != pd_idx) {
off_srcs[count] = dev->offset;
xor_srcs[count++] = dev->page;
}
}
}
/* 1/ if we prexor'd then the dest is reused as a source *2/ifwedidnotprexorthenweareredoingtheparity *setASYNC_TX_XOR_DROP_DSTandASYNC_TX_XOR_ZERO_DST *forthesynchronousxorcase
*/
last_stripe = !head_sh->batch_head ||
list_first_entry(&sh->batch_list, struct stripe_head, batch_list) == head_sh; if (last_stripe) {
flags = ASYNC_TX_ACK |
(prexor ? ASYNC_TX_XOR_DROP_DST : ASYNC_TX_XOR_ZERO_DST);
for (i = 0; i < sh->disks; i++) { if (sh->pd_idx == i || sh->qd_idx == i) continue; if (!test_bit(R5_Discard, &sh->dev[i].flags)) break;
} if (i >= sh->disks) {
atomic_inc(&sh->count);
set_bit(R5_Discard, &sh->dev[sh->pd_idx].flags);
set_bit(R5_Discard, &sh->dev[sh->qd_idx].flags);
ops_complete_reconstruct(sh); return;
}
if (test_bit(STRIPE_OP_COMPUTE_BLK, &ops_request)) { if (level < 6)
tx = ops_run_compute5(sh, percpu); else { if (sh->ops.target2 < 0 || sh->ops.target < 0)
tx = ops_run_compute6_1(sh, percpu); else
tx = ops_run_compute6_2(sh, percpu);
} /* terminate the chain if reconstruct is not set to be run */ if (tx && !test_bit(STRIPE_OP_RECONSTRUCT, &ops_request))
async_tx_ack(tx);
}
if (test_bit(STRIPE_OP_PREXOR, &ops_request)) { if (level < 6)
tx = ops_run_prexor5(sh, percpu, tx); else
tx = ops_run_prexor6(sh, percpu, tx);
}
if (test_bit(STRIPE_OP_PARTIAL_PARITY, &ops_request))
tx = ops_run_partial_parity(sh, percpu, tx);
if (test_bit(STRIPE_OP_BIODRAIN, &ops_request)) {
tx = ops_run_biodrain(sh, tx);
overlap_clear++;
}
if (test_bit(STRIPE_OP_RECONSTRUCT, &ops_request)) { if (level < 6)
ops_run_reconstruct5(sh, percpu, tx); else
ops_run_reconstruct6(sh, percpu, tx);
}
sh = alloc_stripe(conf->slab_cache, gfp, conf->pool_size, conf); if (!sh) return0;
if (grow_buffers(sh, gfp)) {
shrink_buffers(sh);
free_stripe(conf->slab_cache, sh); return0;
}
sh->hash_lock_index =
conf->max_nr_stripes % NR_STRIPE_HASH_LOCKS; /* we just created an active stripe so... */
atomic_inc(&conf->active_stripes);
/* Step 4, return new stripes to service */ while(!list_empty(&newstripes)) {
nsh = list_entry(newstripes.next, struct stripe_head, lru);
list_del_init(&nsh->lru);
#if PAGE_SIZE != DEFAULT_STRIPE_SIZE for (i = 0; i < nsh->nr_pages; i++) { if (nsh->pages[i]) continue;
nsh->pages[i] = alloc_page(GFP_NOIO); if (!nsh->pages[i])
err = -ENOMEM;
}
for (i = conf->raid_disks; i < newsize; i++) { if (nsh->dev[i].page) continue;
nsh->dev[i].page = raid5_get_dev_page(nsh, i);
nsh->dev[i].orig_page = nsh->dev[i].page;
nsh->dev[i].offset = raid5_get_page_offset(nsh, i);
} #else for (i=conf->raid_disks; i < newsize; i++) if (nsh->dev[i].page == NULL) { struct page *p = alloc_page(GFP_NOIO);
nsh->dev[i].page = p;
nsh->dev[i].orig_page = p;
nsh->dev[i].offset = 0; if (!p)
err = -ENOMEM;
} #endif
raid5_release_stripe(nsh);
} /* critical section pass, GFP_NOIO no longer needed */
if (!err)
conf->pool_size = newsize;
mutex_unlock(&conf->cache_size_mutex);
for (i=0 ; i<disks; i++) if (bi == &sh->dev[i].req) break;
pr_debug("end_read_request %llu/%d, count: %d, error %d.\n",
(unsignedlonglong)sh->sector, i, atomic_read(&sh->count),
bi->bi_status); if (i == disks) {
BUG(); return;
} if (test_bit(R5_ReadRepl, &sh->dev[i].flags)) /* If replacement finished while this request was outstanding, *'replacement'mightbeNULLalready. *Inthatcaseitmoveddownto'rdev'. *rdevisnotremoveduntilallrequestsarefinished.
*/
rdev = conf->disks[i].replacement; if (!rdev)
rdev = conf->disks[i].rdev;
if (use_new_offset(conf, sh))
s = sh->sector + rdev->new_data_offset; else
s = sh->sector + rdev->data_offset; if (!bi->bi_status) {
set_bit(R5_UPTODATE, &sh->dev[i].flags); if (test_bit(R5_ReadError, &sh->dev[i].flags)) { /* Note that this cannot happen on a *replacementdevice.Wejustfailthoseon *anyerror
*/
pr_info_ratelimited( "md/raid:%s: read error corrected (%lu sectors at %llu on %pg)\n",
mdname(conf->mddev), RAID5_STRIPE_SECTORS(conf),
(unsignedlonglong)s,
rdev->bdev);
atomic_add(RAID5_STRIPE_SECTORS(conf), &rdev->corrected_errors);
clear_bit(R5_ReadError, &sh->dev[i].flags);
clear_bit(R5_ReWrite, &sh->dev[i].flags);
} elseif (test_bit(R5_ReadNoMerge, &sh->dev[i].flags))
clear_bit(R5_ReadNoMerge, &sh->dev[i].flags);
if (test_bit(R5_InJournal, &sh->dev[i].flags)) /* *endreadforapageinjournal,this *mustbepreparingforprexorinrmw
*/
set_bit(R5_OrigPageUPTDODATE, &sh->dev[i].flags);
if (atomic_read(&rdev->read_errors))
atomic_set(&rdev->read_errors, 0);
} else { int retry = 0; int set_bad = 0;
case ALGORITHM_ROTATING_ZERO_RESTART: /* Exactly the same as RIGHT_ASYMMETRIC, but or *ofblocksforcomputingQisdifferent.
*/
pd_idx = sector_div(stripe2, raid_disks);
qd_idx = pd_idx + 1; if (pd_idx == raid_disks-1) {
(*dd_idx)++; /* Q D D D P */
qd_idx = 0;
} elseif (*dd_idx >= pd_idx)
(*dd_idx) += 2; /* D D P Q D */
ddf_layout = 1; break;
case ALGORITHM_ROTATING_N_RESTART: /* Same a left_asymmetric, by first stripe is *DDDPQratherthan *QDDDP
*/
stripe2 += 1;
pd_idx = raid_disks - 1 - sector_div(stripe2, raid_disks);
qd_idx = pd_idx + 1; if (pd_idx == raid_disks-1) {
(*dd_idx)++; /* Q D D D P */
qd_idx = 0;
} elseif (*dd_idx >= pd_idx)
(*dd_idx) += 2; /* D D P Q D */
ddf_layout = 1; break;
case ALGORITHM_ROTATING_N_CONTINUE: /* Same as left_symmetric but Q is before P */
pd_idx = raid_disks - 1 - sector_div(stripe2, raid_disks);
qd_idx = (pd_idx + raid_disks - 1) % raid_disks;
*dd_idx = (pd_idx + 1 + *dd_idx) % raid_disks;
ddf_layout = 1; break;
case ALGORITHM_LEFT_ASYMMETRIC_6: /* RAID5 left_asymmetric, with Q on last device */
pd_idx = data_disks - sector_div(stripe2, raid_disks-1); if (*dd_idx >= pd_idx)
(*dd_idx)++;
qd_idx = raid_disks - 1; break;
case ALGORITHM_RIGHT_ASYMMETRIC_6:
pd_idx = sector_div(stripe2, raid_disks-1); if (*dd_idx >= pd_idx)
(*dd_idx)++;
qd_idx = raid_disks - 1; break;
if (i == sh->pd_idx) return0; switch(conf->level) { case4: break; case5: switch (algorithm) { case ALGORITHM_LEFT_ASYMMETRIC: case ALGORITHM_RIGHT_ASYMMETRIC: if (i > sh->pd_idx)
i--; break; case ALGORITHM_LEFT_SYMMETRIC: case ALGORITHM_RIGHT_SYMMETRIC: if (i < sh->pd_idx)
i += raid_disks;
i -= (sh->pd_idx + 1); break; case ALGORITHM_PARITY_0:
i -= 1; break; case ALGORITHM_PARITY_N: break; default:
BUG();
} break; case6: if (i == sh->qd_idx) return0; /* It is the Q disk */ switch (algorithm) { case ALGORITHM_LEFT_ASYMMETRIC: case ALGORITHM_RIGHT_ASYMMETRIC: case ALGORITHM_ROTATING_ZERO_RESTART: case ALGORITHM_ROTATING_N_RESTART: if (sh->pd_idx == raid_disks-1)
i--; /* Q D D D P */ elseif (i > sh->pd_idx)
i -= 2; /* D D P Q D */ break; case ALGORITHM_LEFT_SYMMETRIC: case ALGORITHM_RIGHT_SYMMETRIC: if (sh->pd_idx == raid_disks-1)
i--; /* Q D D D P */ else { /* D D P Q D */ if (i < sh->pd_idx)
i += raid_disks;
i -= (sh->pd_idx + 2);
} break; case ALGORITHM_PARITY_0:
i -= 2; break; case ALGORITHM_PARITY_N: break; case ALGORITHM_ROTATING_N_CONTINUE: /* Like left_symmetric, but P is before Q */ if (sh->pd_idx == 0)
i--; /* P D D D Q */ else { /* D D Q P D */ if (i < sh->pd_idx)
i += raid_disks;
i -= (sh->pd_idx + 1);
} break; case ALGORITHM_LEFT_ASYMMETRIC_6: case ALGORITHM_RIGHT_ASYMMETRIC_6: if (i > sh->pd_idx)
i--; break; case ALGORITHM_LEFT_SYMMETRIC_6: case ALGORITHM_RIGHT_SYMMETRIC_6: if (i < sh->pd_idx)
i += data_disks + 1;
i -= (sh->pd_idx + 1); break; case ALGORITHM_PARITY_0_6:
i -= 1; break; default:
BUG();
} break;
}
staticvoid
schedule_reconstruction(struct stripe_head *sh, struct stripe_head_state *s, int rcw, int expand)
{ int i, pd_idx = sh->pd_idx, qd_idx = sh->qd_idx, disks = sh->disks; struct r5conf *conf = sh->raid_conf; int level = conf->level;
if (rcw) { /* *Insomecases,handle_stripe_dirtyinginitiallydecidedto *runrmwandallocatesextrapageforprexor.However,rcwis *cheaperlateron.Weneedtofreetheextrapagenow, *becausewewon'tbeabletodothatinops_complete_prexor().
*/
r5c_release_extra_page(sh);
for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i];
if (dev->towrite && !delay_towrite(conf, dev, s)) {
set_bit(R5_LOCKED, &dev->flags);
set_bit(R5_Wantdrain, &dev->flags); if (!expand)
clear_bit(R5_UPTODATE, &dev->flags);
s->locked++;
} elseif (test_bit(R5_InJournal, &dev->flags)) {
set_bit(R5_LOCKED, &dev->flags);
s->locked++;
}
} /* if we are not expanding this is a proper write request, and *therewillbebioswithnewdatatobedrainedintothe *stripecache
*/ if (!expand) { if (!s->locked) /* False alarm, nothing to do */ return;
sh->reconstruct_state = reconstruct_state_drain_run;
set_bit(STRIPE_OP_BIODRAIN, &s->ops_request);
} else
sh->reconstruct_state = reconstruct_state_run;
staticbool stripe_bio_overlaps(struct stripe_head *sh, struct bio *bi, int dd_idx, int forwrite)
{ struct r5conf *conf = sh->raid_conf; struct bio **bip;
pr_debug("checking bi b#%llu to stripe s#%llu\n",
bi->bi_iter.bi_sector, sh->sector);
/* Don't allow new IO added to stripes in batch list */ if (sh->batch_head) returntrue;
if (forwrite)
bip = &sh->dev[dd_idx].towrite; else
bip = &sh->dev[dd_idx].toread;
while (*bip && (*bip)->bi_iter.bi_sector < bi->bi_iter.bi_sector) { if (bio_end_sector(*bip) > bi->bi_iter.bi_sector) returntrue;
bip = &(*bip)->bi_next;
}
if (*bip && (*bip)->bi_iter.bi_sector < bio_end_sector(bi)) returntrue;
if (forwrite && raid5_has_ppl(conf)) { /* *WithPPLonlywritestoconsecutivedatachunkswithina *stripeareallowedbecauseforasinglestripe_headwecan *onlyhaveonePPLentryatatime,whichdescribesonedata *range.Notreallyanoverlap,butR5_Overlapcanbe *usedtohandlethis.
*/
sector_t sector;
sector_t first = 0;
sector_t last = 0; int count = 0; int i;
for (i = 0; i < sh->disks; i++) { if (i != sh->pd_idx &&
(i == dd_idx || sh->dev[i].towrite)) {
sector = sh->dev[i].sector; if (count == 0 || sector < first)
first = sector; if (sector > last)
last = sector;
count++;
}
}
staticvoid __add_stripe_bio(struct stripe_head *sh, struct bio *bi, int dd_idx, int forwrite, int previous)
{ struct r5conf *conf = sh->raid_conf; struct bio **bip; int firstwrite = 0;
if (forwrite) {
bip = &sh->dev[dd_idx].towrite; if (!*bip)
firstwrite = 1;
} else {
bip = &sh->dev[dd_idx].toread;
}
while (*bip && (*bip)->bi_iter.bi_sector < bi->bi_iter.bi_sector)
bip = &(*bip)->bi_next;
if (!forwrite || previous)
clear_bit(STRIPE_BATCH_READY, &sh->state);
/* *Eachstripe/devcanhaveoneormorebiosattached. *toread/towritepointtothefirstinachain. *Thebi_nextchainmustbeinorder.
*/ staticbool add_stripe_bio(struct stripe_head *sh, struct bio *bi, int dd_idx, int forwrite, int previous)
{
spin_lock_irq(&sh->stripe_lock);
staticvoid
handle_failed_stripe(struct r5conf *conf, struct stripe_head *sh, struct stripe_head_state *s, int disks)
{ int i;
BUG_ON(sh->batch_head); for (i = disks; i--; ) { struct bio *bi;
if (test_bit(R5_ReadError, &sh->dev[i].flags)) { struct md_rdev *rdev = conf->disks[i].rdev;
if (rdev && test_bit(In_sync, &rdev->flags) &&
!test_bit(Faulty, &rdev->flags))
atomic_inc(&rdev->nr_pending); else
rdev = NULL; if (rdev) { if (!rdev_set_badblocks(
rdev,
sh->sector,
RAID5_STRIPE_SECTORS(conf), 0))
md_error(conf->mddev, rdev);
rdev_dec_pending(rdev, conf->mddev);
}
}
spin_lock_irq(&sh->stripe_lock); /* fail all writes first */
bi = sh->dev[i].towrite;
sh->dev[i].towrite = NULL;
sh->overwrite_disks = 0;
spin_unlock_irq(&sh->stripe_lock);
log_stripe_write_finished(sh);
if (test_and_clear_bit(R5_Overlap, &sh->dev[i].flags))
wake_up_bit(&sh->dev[i].flags, R5_Overlap);
while (bi && bi->bi_iter.bi_sector <
sh->dev[i].sector + RAID5_STRIPE_SECTORS(conf)) { struct bio *nextbi = r5_next_bio(conf, bi, sh->dev[i].sector);
md_write_end(conf->mddev);
bio_io_error(bi);
bi = nextbi;
} /* and fail all 'written' */
bi = sh->dev[i].written;
sh->dev[i].written = NULL; if (test_and_clear_bit(R5_SkipCopy, &sh->dev[i].flags)) {
WARN_ON(test_bit(R5_UPTODATE, &sh->dev[i].flags));
sh->dev[i].page = sh->dev[i].orig_page;
}
while (bi && bi->bi_iter.bi_sector <
sh->dev[i].sector + RAID5_STRIPE_SECTORS(conf)) { struct bio *bi2 = r5_next_bio(conf, bi, sh->dev[i].sector);
md_write_end(conf->mddev);
bio_io_error(bi);
bi = bi2;
}
/* fail any reads if this device is non-operational and *thedatahasnotreachedthecacheyet.
*/ if (!test_bit(R5_Wantfill, &sh->dev[i].flags) &&
s->failed > conf->max_degraded &&
(!test_bit(R5_Insync, &sh->dev[i].flags) ||
test_bit(R5_ReadError, &sh->dev[i].flags))) {
spin_lock_irq(&sh->stripe_lock);
bi = sh->dev[i].toread;
sh->dev[i].toread = NULL;
spin_unlock_irq(&sh->stripe_lock); if (test_and_clear_bit(R5_Overlap, &sh->dev[i].flags))
wake_up_bit(&sh->dev[i].flags, R5_Overlap); if (bi)
s->to_read--; while (bi && bi->bi_iter.bi_sector <
sh->dev[i].sector + RAID5_STRIPE_SECTORS(conf)) { struct bio *nextbi =
r5_next_bio(conf, bi, sh->dev[i].sector);
bio_io_error(bi);
bi = nextbi;
}
} /* If we were in the middle of a write the parity block might *stillbelocked-sojustclearallR5_LOCKEDflags
*/
clear_bit(R5_LOCKED, &sh->dev[i].flags);
}
s->to_write = 0;
s->written = 0;
if (test_and_clear_bit(STRIPE_FULL_WRITE, &sh->state)) if (atomic_dec_and_test(&conf->pending_full_writes))
md_wakeup_thread(conf->mddev->thread);
}
staticvoid
handle_failed_sync(struct r5conf *conf, struct stripe_head *sh, struct stripe_head_state *s)
{ int abort = 0; int i;
BUG_ON(sh->batch_head);
clear_bit(STRIPE_SYNCING, &sh->state); if (test_and_clear_bit(R5_Overlap, &sh->dev[sh->pd_idx].flags))
wake_up_bit(&sh->dev[sh->pd_idx].flags, R5_Overlap);
s->syncing = 0;
s->replacing = 0; /* There is nothing more to do for sync/check/repair. *Don'tevenneedtoabortasthatishandledelsewhere *ifneeded,andnotalwayswantede.g.ifthereisaknown *badblockhere. *Forrecover/replaceweneedtorecordabadblockonall *non-syncdevices,oraborttherecovery
*/ if (test_bit(MD_RECOVERY_RECOVER, &conf->mddev->recovery)) { /* During recovery devices cannot be removed, so *lockingandrefcountingofrdevsisnotneeded
*/ for (i = 0; i < conf->raid_disks; i++) { struct md_rdev *rdev = conf->disks[i].rdev;
if (test_bit(R5_LOCKED, &dev->flags) ||
test_bit(R5_UPTODATE, &dev->flags)) /* No point reading this as we already have it or have *decidedtogetit.
*/ return0;
if (dev->toread ||
(dev->towrite && !test_bit(R5_OVERWRITE, &dev->flags))) /* We need this block to directly satisfy a request */ return1;
if (s->syncing || s->expanding ||
(s->replacing && want_replace(sh, disk_idx))) /* When syncing, or expanding we read everything. *Whenreplacing,weneedthereplacedblock.
*/ return1;
if ((s->failed >= 1 && fdev[0]->toread) ||
(s->failed >= 2 && fdev[1]->toread)) /* If we want to read from a failed device, then *weneedtoactuallyreadeveryotherdevice.
*/ return1;
/* Sometimes neither read-modify-write nor reconstruct-write *cyclescanwork.Inthosecaseswereadeveryblockwe *can.Thentheparity-updateiscertaintohaveenoughto *workwith. *Thiscanonlybeaproblemwhenweneedtowritesomething, *andsomedevicehasfailed.Ifeitherofthosetests *failweneedlooknofurther.
*/ if (!s->failed || !s->to_write) return0;
if (test_bit(R5_Insync, &dev->flags) &&
!test_bit(STRIPE_PREREAD_ACTIVE, &sh->state)) /* Pre-reads at not permitted until after short delay *togathermultiplerequests.Howeverifthis *deviceisnoInsync,theblockcouldonlybecomputed *andthereisnoneedtodelaythat.
*/ return0;
for (i = 0; i < s->failed && i < 2; i++) { if (fdev[i]->towrite &&
!test_bit(R5_UPTODATE, &fdev[i]->flags) &&
!test_bit(R5_OVERWRITE, &fdev[i]->flags)) /* If we have a partial write to a failed *device,thenwewillneedtoreconstruct *thecontentofthatdevice,soallother *devicesmustberead.
*/ return1;
if (s->failed >= 2 &&
(fdev[i]->towrite ||
s->failed_num[i] == sh->pd_idx ||
s->failed_num[i] == sh->qd_idx) &&
!test_bit(R5_UPTODATE, &fdev[i]->flags)) /* In max degraded raid6, If the failed disk is P, Q, *orwewanttoreadthefaileddisk,weneedtodo *reconstruct-write.
*/
force_rcw = true;
}
/* If we are forced to do a reconstruct-write, because parity *cannotbetrustedandwearecurrentlyrecoveringit,there *isextraneedtobecareful. *Ifoneofthedevicesthatwewouldneedtoread,because *itisnotbeingoverwritten(andmaybenotwrittenatall) *ismissing/faulty,thenweneedtoreadeverythingwecan.
*/ if (!force_rcw &&
sh->sector < sh->raid_conf->mddev->resync_offset) /* reconstruct-write isn't being forced */ return0; for (i = 0; i < s->failed && i < 2; i++) { if (s->failed_num[i] != sh->pd_idx &&
s->failed_num[i] != sh->qd_idx &&
!test_bit(R5_UPTODATE, &fdev[i]->flags) &&
!test_bit(R5_OVERWRITE, &fdev[i]->flags)) return1;
}
return0;
}
/* fetch_block - checks the given member device to see if its data needs *tobereadorcomputedtosatisfyarequest. * *Returns1whennomorememberdevicesneedtobechecked,otherwisereturns *0totelltheloopinhandle_stripe_filltocontinue
*/ staticint fetch_block(struct stripe_head *sh, struct stripe_head_state *s, int disk_idx, int disks)
{ struct r5dev *dev = &sh->dev[disk_idx];
/* is the data in this block needed, and can we get it? */ if (need_this_block(sh, s, disk_idx, disks)) { /* we would like to get this block, possibly by computing it, *otherwisereaditifthebackingdiskisinsync
*/
BUG_ON(test_bit(R5_Wantcompute, &dev->flags));
BUG_ON(test_bit(R5_Wantread, &dev->flags));
BUG_ON(sh->batch_head);
if ((s->uptodate == disks - 1) &&
((sh->qd_idx >= 0 && sh->pd_idx == disk_idx) ||
(s->failed && (disk_idx == s->failed_num[0] ||
disk_idx == s->failed_num[1])))) { /* have disk failed, and we're requested to fetch it; *docomputeit
*/
pr_debug("Computing stripe %llu block %d\n",
(unsignedlonglong)sh->sector, disk_idx);
set_bit(STRIPE_COMPUTE_RUN, &sh->state);
set_bit(STRIPE_OP_COMPUTE_BLK, &s->ops_request);
set_bit(R5_Wantcompute, &dev->flags);
sh->ops.target = disk_idx;
sh->ops.target2 = -1; /* no 2nd target */
s->req_compute = 1; /* Careful: from this point on 'uptodate' is in the eye *ofraid_run_opswhichservices'compute'operations *beforewrites.R5_Wantcomputeflagsablockthatwill *beR5_UPTODATEbythetimeitisneededfora *subsequentoperation.
*/
s->uptodate++; return1;
} elseif (s->uptodate == disks-2 && s->failed >= 2) { /* Computing 2-failure is *very* expensive; only *doitiffailed>=2
*/ int other; for (other = disks; other--; ) { if (other == disk_idx) continue; if (!test_bit(R5_UPTODATE,
&sh->dev[other].flags)) break;
}
BUG_ON(other < 0);
pr_debug("Computing stripe %llu blocks %d,%d\n",
(unsignedlonglong)sh->sector,
disk_idx, other);
set_bit(STRIPE_COMPUTE_RUN, &sh->state);
set_bit(STRIPE_OP_COMPUTE_BLK, &s->ops_request);
set_bit(R5_Wantcompute, &sh->dev[disk_idx].flags);
set_bit(R5_Wantcompute, &sh->dev[other].flags);
sh->ops.target = disk_idx;
sh->ops.target2 = other;
s->uptodate += 2;
s->req_compute = 1; return1;
} elseif (test_bit(R5_Insync, &dev->flags)) {
set_bit(R5_LOCKED, &dev->flags);
set_bit(R5_Wantread, &dev->flags);
s->locked++;
pr_debug("Reading block %d (sync=%d)\n",
disk_idx, s->syncing);
}
}
return0;
}
/* *handle_stripe_fill-readorcomputedatatosatisfypendingrequests.
*/ staticvoid handle_stripe_fill(struct stripe_head *sh, struct stripe_head_state *s, int disks)
{ int i;
/* look for blocks to read/compute, skip this if a compute *isalreadyinflight,orifthestripecontentsareinthe *midstofchangingduetoawrite
*/ if (!test_bit(STRIPE_COMPUTE_RUN, &sh->state) && !sh->check_state &&
!sh->reconstruct_state) {
for (i = disks; i--; ) if (sh->dev[i].written) {
dev = &sh->dev[i]; if (!test_bit(R5_LOCKED, &dev->flags) &&
(test_bit(R5_UPTODATE, &dev->flags) ||
test_bit(R5_Discard, &dev->flags) ||
test_bit(R5_SkipCopy, &dev->flags))) { /* We can return any write requests */ struct bio *wbi, *wbi2;
pr_debug("Return write for disc %d\n", i); if (test_and_clear_bit(R5_Discard, &dev->flags))
clear_bit(R5_UPTODATE, &dev->flags); if (test_and_clear_bit(R5_SkipCopy, &dev->flags)) {
WARN_ON(test_bit(R5_UPTODATE, &dev->flags));
}
do_endio = true;
/* Check whether resync is now happening or should start. *Ifyes,thenthearrayisdirty(afteruncleanshutdownor *initialcreation),soparityinsomestripesmightbeinconsistent. *Inthiscase,weneedtoalwaysdoreconstruct-write,toensure *thatincaseofdrivefailureorread-errorcorrection,we *generatecorrectdatafromtheparity.
*/ if (conf->rmw_level == PARITY_DISABLE_RMW ||
(resync_offset < MaxSector && sh->sector >= resync_offset &&
s->failed == 0)) { /* Calculate the real rcw later - for now make it *looklikercwischeaper
*/
rcw = 1; rmw = 2;
pr_debug("force RCW rmw_level=%u, resync_offset=%llu sh->sector=%llu\n",
conf->rmw_level, (unsignedlonglong)resync_offset,
(unsignedlonglong)sh->sector);
} elsefor (i = disks; i--; ) { /* would I have to read this buffer for read_modify_write */ struct r5dev *dev = &sh->dev[i]; if (((dev->towrite && !delay_towrite(conf, dev, s)) ||
i == sh->pd_idx || i == sh->qd_idx ||
test_bit(R5_InJournal, &dev->flags)) &&
!test_bit(R5_LOCKED, &dev->flags) &&
!(uptodate_for_rmw(dev) ||
test_bit(R5_Wantcompute, &dev->flags))) { if (test_bit(R5_Insync, &dev->flags))
rmw++; else
rmw += 2*disks; /* cannot read it */
} /* Would I have to read this buffer for reconstruct_write */ if (!test_bit(R5_OVERWRITE, &dev->flags) &&
i != sh->pd_idx && i != sh->qd_idx &&
!test_bit(R5_LOCKED, &dev->flags) &&
!(test_bit(R5_UPTODATE, &dev->flags) ||
test_bit(R5_Wantcompute, &dev->flags))) { if (test_bit(R5_Insync, &dev->flags))
rcw++; else
rcw += 2*disks;
}
}
pr_debug("for sector %llu state 0x%lx, rmw=%d rcw=%d\n",
(unsignedlonglong)sh->sector, sh->state, rmw, rcw);
set_bit(STRIPE_HANDLE, &sh->state); if ((rmw < rcw || (rmw == rcw && conf->rmw_level == PARITY_PREFER_RMW)) && rmw > 0) { /* prefer read-modify-write, but need to get some data */
mddev_add_trace_msg(conf->mddev, "raid5 rmw %llu %d",
sh->sector, rmw);
for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i]; if (test_bit(R5_InJournal, &dev->flags) &&
dev->page == dev->orig_page &&
!test_bit(R5_LOCKED, &sh->dev[sh->pd_idx].flags)) { /* alloc page for prexor */ struct page *p = alloc_page(GFP_NOIO);
/* now if nothing is locked, and if we have enough data, *wecanstartawriterequest
*/ /* since handle_stripe can be called at any time we need to handle the *casewhereacomputeblockoperationhasbeensubmittedandthena *subsequentcallwantstostartawriterequest.raid_run_opsonly *handlesthecasewherecomputeblockandreconstructarerequested *simultaneously.Ifthisisnotthecasethennewwritesneedtobe *heldoffuntilthecomputecompletes.
*/ if ((s->req_compute || !test_bit(STRIPE_COMPUTE_RUN, &sh->state)) &&
(s->locked == 0 && (rcw == 0 || rmw == 0) &&
!test_bit(STRIPE_BIT_DELAY, &sh->state)))
schedule_reconstruction(sh, s, rcw == 0, 0); return0;
}
switch (sh->check_state) { case check_state_idle: /* start a new check operation if there are no failures */ if (s->failed == 0) {
BUG_ON(s->uptodate != disks);
sh->check_state = check_state_run;
set_bit(STRIPE_OP_CHECK, &s->ops_request);
clear_bit(R5_UPTODATE, &sh->dev[sh->pd_idx].flags);
s->uptodate--; break;
}
dev = &sh->dev[s->failed_num[0]];
fallthrough; case check_state_compute_result:
sh->check_state = check_state_idle; if (!dev)
dev = &sh->dev[sh->pd_idx];
/* check that a write has not made the stripe insync */ if (test_bit(STRIPE_INSYNC, &sh->state)) break;
/* either failed parity check, or recovery is happening */
BUG_ON(!test_bit(R5_UPTODATE, &dev->flags));
BUG_ON(s->uptodate != disks);
set_bit(STRIPE_INSYNC, &sh->state); break; case check_state_run: break; /* we will be called again upon completion */ case check_state_check_result:
sh->check_state = check_state_idle;
/* if a failure occurred during the check operation, leave *STRIPE_INSYNCnotsetandletthestripebehandledagain
*/ if (s->failed) break;
/* handle a successful check operation, if parity is correct *wearedone.Otherwiseupdatethemismatchcountandrepair *parityif!MD_RECOVERY_CHECK
*/ if ((sh->ops.zero_sum_result & SUM_CHECK_P_RESULT) == 0) /* parity is correct (on disc, *notinbufferanymore)
*/
set_bit(STRIPE_INSYNC, &sh->state); else {
atomic64_add(RAID5_STRIPE_SECTORS(conf), &conf->mddev->resync_mismatches); if (test_bit(MD_RECOVERY_CHECK, &conf->mddev->recovery)) { /* don't try to repair!! */
set_bit(STRIPE_INSYNC, &sh->state);
pr_warn_ratelimited("%s: mismatch sector in range " "%llu-%llu\n", mdname(conf->mddev),
(unsignedlonglong) sh->sector,
(unsignedlonglong) sh->sector +
RAID5_STRIPE_SECTORS(conf));
} else {
sh->check_state = check_state_compute_run;
set_bit(STRIPE_COMPUTE_RUN, &sh->state);
set_bit(STRIPE_OP_COMPUTE_BLK, &s->ops_request);
set_bit(R5_Wantcompute,
&sh->dev[sh->pd_idx].flags);
sh->ops.target = sh->pd_idx;
sh->ops.target2 = -1;
s->uptodate++;
}
} break; case check_state_compute_run: break; default:
pr_err("%s: unknown check_state: %d sector: %llu\n",
__func__, sh->check_state,
(unsignedlonglong) sh->sector);
BUG();
}
}
staticvoid handle_parity_checks6(struct r5conf *conf, struct stripe_head *sh, struct stripe_head_state *s, int disks)
{ int pd_idx = sh->pd_idx; int qd_idx = sh->qd_idx; struct r5dev *dev;
/* Want to check and possibly repair P and Q. *Howevertherecouldbeone'failed'device,inwhich *casewecanonlycheckoneofthem,possiblyusingthe *othertogeneratemissingdata
*/
switch (sh->check_state) { case check_state_idle: /* start a new check operation if there are < 2 failures */ if (s->failed == s->q_failed) { /* The only possible failed device holds Q, so it *makessensetocheckP(Ifanythingelsewerefailed, *wewouldhaveusedPtorecreateit).
*/
sh->check_state = check_state_run;
} if (!s->q_failed && s->failed < 2) { /* Q is not failed, and we didn't use it to generate *anything,soitmakessensetocheckit
*/ if (sh->check_state == check_state_run)
sh->check_state = check_state_run_pq; else
sh->check_state = check_state_run_q;
}
/* discard potentially stale zero_sum_result */
sh->ops.zero_sum_result = 0;
if (sh->check_state == check_state_run) { /* async_xor_zero_sum destroys the contents of P */
clear_bit(R5_UPTODATE, &sh->dev[pd_idx].flags);
s->uptodate--;
} if (sh->check_state >= check_state_run &&
sh->check_state <= check_state_run_pq) { /* async_syndrome_zero_sum preserves P and Q, so *noneedtomarkthem!uptodatehere
*/
set_bit(STRIPE_OP_CHECK, &s->ops_request); break;
}
/* we have 2-disk failure */
BUG_ON(s->failed != 2);
fallthrough; case check_state_compute_result:
sh->check_state = check_state_idle;
/* check that a write has not made the stripe insync */ if (test_bit(STRIPE_INSYNC, &sh->state)) break;
/* now write out any block on a failed drive, *orPorQiftheywererecomputed
*/
dev = NULL; if (s->failed == 2) {
dev = &sh->dev[s->failed_num[1]];
s->locked++;
set_bit(R5_LOCKED, &dev->flags);
set_bit(R5_Wantwrite, &dev->flags);
} if (s->failed >= 1) {
dev = &sh->dev[s->failed_num[0]];
s->locked++;
set_bit(R5_LOCKED, &dev->flags);
set_bit(R5_Wantwrite, &dev->flags);
} if (sh->ops.zero_sum_result & SUM_CHECK_P_RESULT) {
dev = &sh->dev[pd_idx];
s->locked++;
set_bit(R5_LOCKED, &dev->flags);
set_bit(R5_Wantwrite, &dev->flags);
} if (sh->ops.zero_sum_result & SUM_CHECK_Q_RESULT) {
dev = &sh->dev[qd_idx];
s->locked++;
set_bit(R5_LOCKED, &dev->flags);
set_bit(R5_Wantwrite, &dev->flags);
} if (WARN_ONCE(dev && !test_bit(R5_UPTODATE, &dev->flags), "%s: disk%td not up to date\n",
mdname(conf->mddev),
dev - (struct r5dev *) &sh->dev)) {
clear_bit(R5_LOCKED, &dev->flags);
clear_bit(R5_Wantwrite, &dev->flags);
s->locked--;
}
set_bit(STRIPE_INSYNC, &sh->state); break; case check_state_run: case check_state_run_q: case check_state_run_pq: break; /* we will be called again upon completion */ case check_state_check_result:
sh->check_state = check_state_idle;
/* handle a successful check operation, if parity is correct *wearedone.Otherwiseupdatethemismatchcountandrepair *parityif!MD_RECOVERY_CHECK
*/ if (sh->ops.zero_sum_result == 0) { /* both parities are correct */ if (!s->failed)
set_bit(STRIPE_INSYNC, &sh->state); else { /* in contrast to the raid5 case we can validate *parity,butstillhaveafailuretowrite *back
*/
sh->check_state = check_state_compute_result; /* Returning at this point means that we may go *offandbringpand/orquptodateagainso *wemakesuretocheckzero_sum_resultagain *toverifyifporqneedwriteback
*/
}
} else {
atomic64_add(RAID5_STRIPE_SECTORS(conf), &conf->mddev->resync_mismatches); if (test_bit(MD_RECOVERY_CHECK, &conf->mddev->recovery)) { /* don't try to repair!! */
set_bit(STRIPE_INSYNC, &sh->state);
pr_warn_ratelimited("%s: mismatch sector in range " "%llu-%llu\n", mdname(conf->mddev),
(unsignedlonglong) sh->sector,
(unsignedlonglong) sh->sector +
RAID5_STRIPE_SECTORS(conf));
} else { int *target = &sh->ops.target;
staticvoid handle_stripe_expansion(struct r5conf *conf, struct stripe_head *sh)
{ int i;
/* We have read all the blocks in this stripe and now we need to *copysomeofthemintoatargetstripeforexpand.
*/ struct dma_async_tx_descriptor *tx = NULL;
BUG_ON(sh->batch_head);
clear_bit(STRIPE_EXPAND_SOURCE, &sh->state); for (i = 0; i < sh->disks; i++) if (i != sh->pd_idx && i != sh->qd_idx) { int dd_idx, j; struct stripe_head *sh2; struct async_submit_ctl submit;
sector_t bn = raid5_compute_blocknr(sh, i, 1);
sector_t s = raid5_compute_sector(conf, bn, 0,
&dd_idx, NULL);
sh2 = raid5_get_active_stripe(conf, NULL, s,
R5_GAS_NOBLOCK | R5_GAS_NOQUIESCE); if (sh2 == NULL) /* so far only the early blocks of this stripe *havebeenrequested.Whenlaterblocks *getrequested,wewilltryagain
*/ continue; if (!test_bit(STRIPE_EXPANDING, &sh2->state) ||
test_bit(R5_Expanded, &sh2->dev[dd_idx].flags)) { /* must have already done this block */
raid5_release_stripe(sh2); continue;
}
/* place all the copies on one channel */
init_async_submit(&submit, 0, tx, NULL, NULL, NULL);
tx = async_memcpy(sh2->dev[dd_idx].page,
sh->dev[i].page, sh2->dev[dd_idx].offset,
sh->dev[i].offset, RAID5_STRIPE_SIZE(conf),
&submit);
/* Now to look around and see what can be done */ for (i=disks; i--; ) { struct md_rdev *rdev; int is_bad = 0;
dev = &sh->dev[i];
pr_debug("check %d: state 0x%lx read %p write %p written %p\n",
i, dev->flags,
dev->toread, dev->towrite, dev->written); /* maybe we can reply to a read * *newwantfillrequestsareonlypermittedwhile *ops_complete_biofillisguaranteedtobeinactive
*/ if (test_bit(R5_UPTODATE, &dev->flags) && dev->toread &&
!test_bit(STRIPE_BIOFILL_RUN, &sh->state))
set_bit(R5_Wantfill, &dev->flags);
/* now count some things */ if (test_bit(R5_LOCKED, &dev->flags))
s->locked++; if (test_bit(R5_UPTODATE, &dev->flags))
s->uptodate++; if (test_bit(R5_Wantcompute, &dev->flags)) {
s->compute++;
BUG_ON(s->compute > 2);
}
if (test_bit(R5_Wantfill, &dev->flags))
s->to_fill++; elseif (dev->toread)
s->to_read++; if (dev->towrite) {
s->to_write++; if (!test_bit(R5_OVERWRITE, &dev->flags))
s->non_overwrite++;
} if (dev->written)
s->written++; /* Prefer to use the replacement for reads, but only *ifitisrecoveredenoughandhasnobadblocks.
*/
rdev = conf->disks[i].replacement; if (rdev && !test_bit(Faulty, &rdev->flags) &&
rdev->recovery_offset >= sh->sector + RAID5_STRIPE_SECTORS(conf) &&
!rdev_has_badblock(rdev, sh->sector,
RAID5_STRIPE_SECTORS(conf)))
set_bit(R5_ReadRepl, &dev->flags); else { if (rdev && !test_bit(Faulty, &rdev->flags))
set_bit(R5_NeedReplace, &dev->flags); else
clear_bit(R5_NeedReplace, &dev->flags);
rdev = conf->disks[i].rdev;
clear_bit(R5_ReadRepl, &dev->flags);
} if (rdev && test_bit(Faulty, &rdev->flags))
rdev = NULL; if (rdev) {
is_bad = rdev_has_badblock(rdev, sh->sector,
RAID5_STRIPE_SECTORS(conf)); if (s->blocked_rdev == NULL) { if (is_bad < 0)
set_bit(BlockedBadBlocks, &rdev->flags); if (rdev_blocked(rdev)) {
s->blocked_rdev = rdev;
atomic_inc(&rdev->nr_pending);
}
}
}
clear_bit(R5_Insync, &dev->flags); if (!rdev) /* Not in-sync */; elseif (is_bad) { /* also not in-sync */ if (!test_bit(WriteErrorSeen, &rdev->flags) &&
test_bit(R5_UPTODATE, &dev->flags)) { /* treat as in-sync, but with a read error *whichwecannowtrytocorrect
*/
set_bit(R5_Insync, &dev->flags);
set_bit(R5_ReadError, &dev->flags);
}
} elseif (test_bit(In_sync, &rdev->flags))
set_bit(R5_Insync, &dev->flags); elseif (sh->sector + RAID5_STRIPE_SECTORS(conf) <= rdev->recovery_offset) /* in sync if before recovery_offset */
set_bit(R5_Insync, &dev->flags); elseif (test_bit(R5_UPTODATE, &dev->flags) &&
test_bit(R5_Expanded, &dev->flags)) /* If we've reshaped into here, we assume it is Insync. *Wewillshortlyupdaterecovery_offsettomake *itofficial.
*/
set_bit(R5_Insync, &dev->flags);
if (test_bit(R5_WriteError, &dev->flags)) { /* This flag does not apply to '.replacement'
* only to .rdev, so make sure to check that*/ struct md_rdev *rdev2 = conf->disks[i].rdev;
if (rdev2 == rdev)
clear_bit(R5_Insync, &dev->flags); if (rdev2 && !test_bit(Faulty, &rdev2->flags)) {
s->handle_bad_blocks = 1;
atomic_inc(&rdev2->nr_pending);
} else
clear_bit(R5_WriteError, &dev->flags);
} if (test_bit(R5_MadeGood, &dev->flags)) { /* This flag does not apply to '.replacement'
* only to .rdev, so make sure to check that*/ struct md_rdev *rdev2 = conf->disks[i].rdev;
sh->check_state = head_sh->check_state;
sh->reconstruct_state = head_sh->reconstruct_state;
spin_lock_irq(&sh->stripe_lock);
sh->batch_head = NULL;
spin_unlock_irq(&sh->stripe_lock); for (i = 0; i < sh->disks; i++) { if (test_and_clear_bit(R5_Overlap, &sh->dev[i].flags))
wake_up_bit(&sh->dev[i].flags, R5_Overlap);
sh->dev[i].flags = head_sh->dev[i].flags &
(~((1 << R5_WriteError) | (1 << R5_Overlap)));
} if (handle_flags == 0 ||
sh->state & handle_flags)
set_bit(STRIPE_HANDLE, &sh->state);
raid5_release_stripe(sh);
}
spin_lock_irq(&head_sh->stripe_lock);
head_sh->batch_head = NULL;
spin_unlock_irq(&head_sh->stripe_lock); for (i = 0; i < head_sh->disks; i++) if (test_and_clear_bit(R5_Overlap, &head_sh->dev[i].flags))
wake_up_bit(&head_sh->dev[i].flags, R5_Overlap); if (head_sh->state & handle_flags)
set_bit(STRIPE_HANDLE, &head_sh->state);
}
staticvoid handle_stripe(struct stripe_head *sh)
{ struct stripe_head_state s; struct r5conf *conf = sh->raid_conf; int i; int prexor; int disks = sh->disks; struct r5dev *pdev, *qdev;
clear_bit(STRIPE_HANDLE, &sh->state);
/* *handle_stripeshouldnotcontinuehandlethebatchedstripe,only *theheadofbatchlistorlonestripecancontinue.Otherwisewe *couldseebreak_stripe_batch_listwarnsabouttheSTRIPE_ACTIVE *issetforthebatchedstripe.
*/ if (clear_batch_ready(sh)) return;
if (test_and_set_bit_lock(STRIPE_ACTIVE, &sh->state)) { /* already being handled, ensure it gets handled
* again when current action finishes */
set_bit(STRIPE_HANDLE, &sh->state); return;
}
if (test_and_clear_bit(STRIPE_BATCH_ERR, &sh->state))
break_stripe_batch_list(sh, 0);
/* Now we check to see if any write operations have recently *completed
*/
prexor = 0; if (sh->reconstruct_state == reconstruct_state_prexor_drain_result)
prexor = 1; if (sh->reconstruct_state == reconstruct_state_drain_result ||
sh->reconstruct_state == reconstruct_state_prexor_drain_result) {
sh->reconstruct_state = reconstruct_state_idle;
/* All the 'written' buffers and the parity block are ready to *bewrittenbacktodisk
*/
BUG_ON(!test_bit(R5_UPTODATE, &sh->dev[sh->pd_idx].flags) &&
!test_bit(R5_Discard, &sh->dev[sh->pd_idx].flags));
BUG_ON(sh->qd_idx >= 0 &&
!test_bit(R5_UPTODATE, &sh->dev[sh->qd_idx].flags) &&
!test_bit(R5_Discard, &sh->dev[sh->qd_idx].flags)); for (i = disks; i--; ) { struct r5dev *dev = &sh->dev[i]; if (test_bit(R5_LOCKED, &dev->flags) &&
(i == sh->pd_idx || i == sh->qd_idx ||
dev->written || test_bit(R5_InJournal,
&dev->flags))) {
pr_debug("Writing block %d\n", i);
set_bit(R5_Wantwrite, &dev->flags); if (prexor) continue; if (s.failed > 1) continue; if (!test_bit(R5_Insync, &dev->flags) ||
((i == sh->pd_idx || i == sh->qd_idx) &&
s.failed == 0))
set_bit(STRIPE_INSYNC, &sh->state);
}
} if (test_and_clear_bit(STRIPE_PREREAD_ACTIVE, &sh->state))
s.dec_preread_active = 1;
}
if (!sh->reconstruct_state && !sh->check_state && !sh->log_io) { if (!r5c_is_writeback(conf->log)) { if (s.to_write)
handle_stripe_dirtying(conf, sh, &s, disks);
} else { /* write back cache */ int ret = 0;
/* First, try handle writes in caching phase */ if (s.to_write)
ret = r5c_try_caching_write(conf, sh, &s,
disks); /* *Ifcachingphasefailed:ret==-EAGAIN *OR *stripeunderreclaim:!caching&&injournal * *fallbacktohandle_stripe_dirtying()
*/ if (ret == -EAGAIN || /* stripe under reclaim: !caching && injournal */
(!test_bit(STRIPE_R5C_CACHING, &sh->state) &&
s.injournal > 0)) {
ret = handle_stripe_dirtying(conf, sh, &s,
disks); if (ret == -EAGAIN) goto finish;
}
}
}
/* maybe we need to check and possibly fix the parity for this stripe *Anyreadswillalreadyhavebeenscheduled,sowejustseeifenough *dataisavailable.Theparitycheckisheldoffwhileparity *dependentoperationsareinflight.
*/ if (sh->check_state ||
(s.syncing && s.locked == 0 &&
!test_bit(STRIPE_COMPUTE_RUN, &sh->state) &&
!test_bit(STRIPE_INSYNC, &sh->state))) { if (conf->level == 6)
handle_parity_checks6(conf, sh, &s, disks); else
handle_parity_checks5(conf, sh, &s, disks);
}
if ((s.replacing || s.syncing) && s.locked == 0
&& !test_bit(STRIPE_COMPUTE_RUN, &sh->state)
&& !test_bit(STRIPE_REPLACED, &sh->state)) { /* Write out to replacement devices where possible */ for (i = 0; i < conf->raid_disks; i++) if (test_bit(R5_NeedReplace, &sh->dev[i].flags)) {
WARN_ON(!test_bit(R5_UPTODATE, &sh->dev[i].flags));
set_bit(R5_WantReplace, &sh->dev[i].flags);
set_bit(R5_LOCKED, &sh->dev[i].flags);
s.locked++;
} if (s.replacing)
set_bit(STRIPE_INSYNC, &sh->state);
set_bit(STRIPE_REPLACED, &sh->state);
} if ((s.syncing || s.replacing) && s.locked == 0 &&
!test_bit(STRIPE_COMPUTE_RUN, &sh->state) &&
test_bit(STRIPE_INSYNC, &sh->state)) {
md_done_sync(conf->mddev, RAID5_STRIPE_SECTORS(conf), 1);
clear_bit(STRIPE_SYNCING, &sh->state); if (test_and_clear_bit(R5_Overlap, &sh->dev[sh->pd_idx].flags))
wake_up_bit(&sh->dev[sh->pd_idx].flags, R5_Overlap);
}
/* If the failed drives are just a ReadError, then we might need *toprogresstherepair/checkprocess
*/ if (s.failed <= conf->max_degraded && !conf->mddev->ro) for (i = 0; i < s.failed; i++) { struct r5dev *dev = &sh->dev[s.failed_num[i]]; if (test_bit(R5_ReadError, &dev->flags)
&& !test_bit(R5_LOCKED, &dev->flags)
&& test_bit(R5_UPTODATE, &dev->flags)
) { if (!test_bit(R5_ReWrite, &dev->flags)) {
set_bit(R5_Wantwrite, &dev->flags);
set_bit(R5_ReWrite, &dev->flags);
} else /* let's read it back */
set_bit(R5_Wantread, &dev->flags);
set_bit(R5_LOCKED, &dev->flags);
s.locked++;
}
}
/* Finish reconstruct operations initiated by the expansion process */ if (sh->reconstruct_state == reconstruct_state_result) { struct stripe_head *sh_src
= raid5_get_active_stripe(conf, NULL, sh->sector,
R5_GAS_PREVIOUS | R5_GAS_NOBLOCK |
R5_GAS_NOQUIESCE); if (sh_src && test_bit(STRIPE_EXPAND_SOURCE, &sh_src->state)) { /* sh cannot be written until sh_src has been read. *soarrangeforshtobedelayedalittle
*/
set_bit(STRIPE_DELAYED, &sh->state);
set_bit(STRIPE_HANDLE, &sh->state); if (!test_and_set_bit(STRIPE_PREREAD_ACTIVE,
&sh_src->state))
atomic_inc(&conf->preread_active_stripes);
raid5_release_stripe(sh_src); goto finish;
} if (sh_src)
raid5_release_stripe(sh_src);
sh->reconstruct_state = reconstruct_state_idle;
clear_bit(STRIPE_EXPANDING, &sh->state); for (i = conf->raid_disks; i--; ) {
set_bit(R5_Wantwrite, &sh->dev[i].flags);
set_bit(R5_LOCKED, &sh->dev[i].flags);
s.locked++;
}
}
if (s.expanded && test_bit(STRIPE_EXPANDING, &sh->state) &&
!sh->reconstruct_state) { /* Need to write out all blocks after computing parity */
sh->disks = conf->raid_disks;
stripe_set_idx(sh->sector, conf, 0, sh);
schedule_reconstruction(sh, &s, 1, 1);
} elseif (s.expanded && !sh->reconstruct_state && s.locked == 0) {
clear_bit(STRIPE_EXPAND_READY, &sh->state);
atomic_dec(&conf->reshape_stripes);
wake_up(&conf->wait_for_reshape);
md_done_sync(conf->mddev, RAID5_STRIPE_SECTORS(conf), 1);
}
finish: /* wait for this device to become unblocked */ if (unlikely(s.blocked_rdev)) { if (conf->mddev->external)
md_wait_for_blocked_rdev(s.blocked_rdev,
conf->mddev); else /* Internal metadata will immediately *bewrittenbyraid5d,sowedon't *needtowaithere.
*/
rdev_dec_pending(s.blocked_rdev,
conf->mddev);
}
if (s.handle_bad_blocks) for (i = disks; i--; ) { struct md_rdev *rdev; struct r5dev *dev = &sh->dev[i]; if (test_and_clear_bit(R5_WriteError, &dev->flags)) { /* We own a safe reference to the rdev */
rdev = conf->disks[i].rdev; if (!rdev_set_badblocks(rdev, sh->sector,
RAID5_STRIPE_SECTORS(conf), 0))
md_error(conf->mddev, rdev);
rdev_dec_pending(rdev, conf->mddev);
} if (test_and_clear_bit(R5_MadeGood, &dev->flags)) {
rdev = conf->disks[i].rdev;
rdev_clear_badblocks(rdev, sh->sector,
RAID5_STRIPE_SECTORS(conf), 0);
rdev_dec_pending(rdev, conf->mddev);
} if (test_and_clear_bit(R5_MadeGoodRepl, &dev->flags)) {
rdev = conf->disks[i].replacement; if (!rdev) /* rdev have been moved down */
rdev = conf->disks[i].rdev;
rdev_clear_badblocks(rdev, sh->sector,
RAID5_STRIPE_SECTORS(conf), 0);
rdev_dec_pending(rdev, conf->mddev);
}
}
if (s.ops_request)
raid_run_ops(sh, s.ops_request);
ops_run_io(sh, &s);
if (s.dec_preread_active) { /* We delay this until after ops_run_io so that if make_request *iswaitingonaflush,itwon'tcontinueuntilthewrites *haveactuallybeensubmitted.
*/
atomic_dec(&conf->preread_active_stripes); if (atomic_read(&conf->preread_active_stripes) <
IO_THRESHOLD)
md_wakeup_thread(conf->mddev->thread);
}
/* No reshape active, so we can trust rdev->data_offset */
align_bio->bi_iter.bi_sector += rdev->data_offset;
did_inc = false; if (conf->quiesce == 0) {
atomic_inc(&conf->active_aligned_reads);
did_inc = true;
} /* need a memory barrier to detect the race with raid5_quiesce() */ if (!did_inc || smp_load_acquire(&conf->quiesce) != 0) { /* quiesce is in progress, so we need to undo io activation and wait *forittofinish
*/ if (did_inc && atomic_dec_and_test(&conf->active_aligned_reads))
wake_up(&conf->wait_for_quiescent);
spin_lock_irq(&conf->device_lock);
wait_event_lock_irq(conf->wait_for_quiescent, conf->quiesce == 0,
conf->device_lock);
atomic_inc(&conf->active_aligned_reads);
spin_unlock_irq(&conf->device_lock);
}
if (cb->list.next == NULL) { int i;
INIT_LIST_HEAD(&cb->list); for (i = 0; i < NR_STRIPE_HASH_LOCKS; i++)
INIT_LIST_HEAD(cb->temp_inactive_list + i);
}
if (!test_and_set_bit(STRIPE_ON_UNPLUG_LIST, &sh->state))
list_add_tail(&sh->lru, &cb->list); else
raid5_release_stripe(sh);
}
if (!range_ahead_of_reshape(mddev, min_sector, max_sector,
conf->reshape_progress)) /* mismatch, need to try again */
ret = true;
spin_unlock_irq(&conf->device_lock);
return ret;
}
staticint add_all_stripe_bios(struct r5conf *conf, struct stripe_request_ctx *ctx, struct stripe_head *sh, struct bio *bi, int forwrite, int previous)
{ int dd_idx;
out_release:
raid5_release_stripe(sh);
out: if (ret == STRIPE_SCHEDULE_AND_RETRY && reshape_interrupted(mddev)) {
bi->bi_status = BLK_STS_RESOURCE;
ret = STRIPE_WAIT_RESHAPE;
pr_err_ratelimited("dm-raid456: io across reshape position while reshape can't make progress");
} return ret;
}
/* *Ifthebiocoversmultipledatadisks,findsectorwithinthebiothathas *thelowestchunkoffsetinthefirstchunk.
*/ static sector_t raid5_bio_lowest_chunk_sector(struct r5conf *conf, struct bio *bi)
{ int sectors_per_chunk = conf->chunk_sectors; int raid_disks = conf->raid_disks; int dd_idx; struct stripe_head sh; unsignedint chunk_offset;
sector_t r_sector = bi->bi_iter.bi_sector & ~((sector_t)RAID5_STRIPE_SECTORS(conf)-1);
sector_t sector;
/* We pass in fake stripe_head to get back parity disk numbers */
sector = raid5_compute_sector(conf, r_sector, 0, &dd_idx, &sh);
chunk_offset = sector_div(sector, sectors_per_chunk); if (sectors_per_chunk - chunk_offset >= bio_sectors(bi)) return r_sector; /* *Biocrossestothenextdatadisk.Checkwhetherit'sinthesame *chunk.
*/
dd_idx++; while (dd_idx == sh.pd_idx || dd_idx == sh.qd_idx)
dd_idx++; if (dd_idx >= raid_disks) return r_sector; return r_sector + sectors_per_chunk - chunk_offset;
}
pr_debug("raid456: %s, logical %llu to %llu\n", __func__,
bi->bi_iter.bi_sector, ctx.last_sector);
/* Bail out if conflicts with reshape and REQ_NOWAIT is set */ if ((bi->bi_opf & REQ_NOWAIT) &&
get_reshape_loc(mddev, conf, logical_sector) == LOC_INSIDE_RESHAPE) {
bio_wouldblock_error(bi); if (rw == WRITE)
md_write_end(mddev); returntrue;
}
md_account_bio(mddev, &bi);
/* We update the metadata at least every 10 seconds, or when *thedataabouttobecopiedwouldover-writethesourceof *thedataatthefrontoftherange.i.e.onenew_stripe *alongfromreshape_progressnew_mapstoafterwhere *reshape_safeold_mapsto
*/
writepos = conf->reshape_progress;
sector_div(writepos, new_data_disks);
readpos = conf->reshape_progress;
sector_div(readpos, data_disks);
safepos = conf->reshape_safe;
sector_div(safepos, data_disks); if (mddev->reshape_backwards) { if (WARN_ON(writepos < reshape_sectors)) return MaxSector;
/* Having calculated the 'writepos' possibly use it *toset'stripe_addr'whichiswherewewillwriteto.
*/ if (mddev->reshape_backwards) { if (WARN_ON(conf->reshape_progress == 0)) return MaxSector;
/* Allow raid5_quiesce to complete */
wait_event(conf->wait_for_reshape, conf->quiesce != 2);
if (test_bit(MD_RECOVERY_RESHAPE, &mddev->recovery)) return reshape_request(mddev, sector_nr, skipped);
/* No need to check resync_max as we never do more than one *stripe,andasresync_maxwillalwaysbeonachunkboundary, *ifthecheckinmd_do_syncdidn'tfire,thereisnochance *ofoversteppingresync_maxhere
*/
/* if there is too many failed drives and we are trying *toresync,thenassertthatwearefinished,becausethereis *nothingwecando.
*/ if (mddev->degraded >= conf->max_degraded &&
test_bit(MD_RECOVERY_SYNC, &mddev->recovery)) {
sector_t rv = mddev->dev_sectors - sector_nr;
*skipped = 1; return rv;
} if (!test_bit(MD_RECOVERY_REQUESTED, &mddev->recovery) &&
!conf->fullsync &&
!mddev->bitmap_ops->start_sync(mddev, sector_nr, &sync_blocks, true) &&
sync_blocks >= RAID5_STRIPE_SECTORS(conf)) { /* we can skip this block, and probably more */
do_div(sync_blocks, RAID5_STRIPE_SECTORS(conf));
*skipped = 1; /* keep things rounded to whole stripes */ return sync_blocks * RAID5_STRIPE_SECTORS(conf);
}
sh = raid5_get_active_stripe(conf, NULL, sector_nr,
R5_GAS_NOBLOCK); if (sh == NULL) {
sh = raid5_get_active_stripe(conf, NULL, sector_nr, 0); /* make sure we don't swamp the stripe cache if someone else *istryingtogetaccess
*/
schedule_timeout_uninterruptible(1);
} /* Need to check if array will still be degraded after recovery/resync *Noteincaseof>1drivefailuresit'spossiblewe'rerebuilding *onedrivewhileleavinganotherfaultydriveinarray.
*/ for (i = 0; i < conf->raid_disks; i++) { struct md_rdev *rdev = conf->disks[i].rdev;
staticint retry_aligned_read(struct r5conf *conf, struct bio *raid_bio, unsignedint offset)
{ /* We may not be able to submit a whole bio at once as there *maynotbeenoughstripe_headsavailable. *Wecannotpre-allocateenoughstripe_headsaswemayneed *morethanexistinthecache(ifwealloweverlargechunks). *Sowedoonestripeheadatatimeandrecordin *->bi_hw_segmentshowmanyhavebeendone. * *We*know*thatthisentireraid_bioisinonechunk,so *itwillbeonlyone'dd_idx'andonlyneedonecalltoraid5_compute_sector.
*/ struct stripe_head *sh; int dd_idx;
sector_t sector, logical_sector, last_sector; int scnt = 0; int handled = 0;
if (scnt < offset) /* already done this stripe */ continue;
sh = raid5_get_active_stripe(conf, NULL, sector,
R5_GAS_NOBLOCK | R5_GAS_NOQUIESCE); if (!sh) { /* failed to get a stripe - must wait */
conf->retry_read_aligned = raid_bio;
conf->retry_read_offset = scnt; return handled;
}
if (batch_size == 0) { for (i = 0; i < NR_STRIPE_HASH_LOCKS; i++) if (!list_empty(temp_inactive_list + i)) break; if (i == NR_STRIPE_HASH_LOCKS) {
spin_unlock_irq(&conf->device_lock);
log_flush_stripe_to_raid(conf);
spin_lock_irq(&conf->device_lock); return batch_size;
}
release_inactive = true;
}
spin_unlock_irq(&conf->device_lock);
blk_start_plug(&plug);
handled = 0;
spin_lock_irq(&conf->device_lock); while (1) { struct bio *bio; int batch_size, released; unsignedint offset;
if (test_bit(MD_SB_CHANGE_PENDING, &mddev->sb_flags)) break;
released = release_stripe_list(conf, conf->temp_inactive_list); if (released)
clear_bit(R5_DID_ALLOC, &conf->cache_state);
if (
!list_empty(&conf->bitmap_list)) { /* Now is a good time to flush some bitmap updates */
conf->seq_flush++;
spin_unlock_irq(&conf->device_lock);
mddev->bitmap_ops->unplug(mddev, true);
spin_lock_irq(&conf->device_lock);
conf->seq_write = conf->seq_flush;
activate_bit_delay(conf, conf->temp_inactive_list);
}
raid5_activate_delayed(conf);
while ((bio = remove_bio_from_retry(conf, &offset))) { int ok;
spin_unlock_irq(&conf->device_lock);
ok = retry_aligned_read(conf, bio, offset);
spin_lock_irq(&conf->device_lock); if (!ok) break;
handled++;
}
spin_unlock_irq(&conf->device_lock); if (test_and_clear_bit(R5_ALLOC_MORE, &conf->cache_state) &&
mutex_trylock(&conf->cache_size_mutex)) {
grow_one_stripe(conf, __GFP_NOWARN); /* Set flag even if allocation failed. This helps *slowdownallocationrequestswhenmemisshort
*/
set_bit(R5_DID_ALLOC, &conf->cache_state);
mutex_unlock(&conf->cache_size_mutex);
}
if (len >= PAGE_SIZE) return -EINVAL; if (kstrtoul(page, 10, &new)) return -EINVAL;
/* *ThevalueshouldnotbebiggerthanPAGE_SIZE.Itrequiresto *bemultipleofDEFAULT_STRIPE_SIZEandthevalueshouldbepower *oftwo.
*/ if (new % DEFAULT_STRIPE_SIZE != 0 || new > PAGE_SIZE || new == 0 || new != roundup_pow_of_two(new)) return -EINVAL;
err = mddev_suspend_and_lock(mddev); if (err) return err;
if (!sectors)
sectors = mddev->dev_sectors; if (!raid_disks) /* size is defined by the smallest of previous and new size */
raid_disks = min(conf->raid_disks, conf->previous_raid_disks);
for (i = 0; i < max_disks; i++) {
conf->disks[i].extra_page = alloc_page(GFP_KERNEL); if (!conf->disks[i].extra_page) goto abort;
}
ret = bioset_init(&conf->bio_split, BIO_POOL_SIZE, 0, 0); if (ret) goto abort;
conf->mddev = mddev;
ret = -ENOMEM;
conf->stripe_hashtbl = kzalloc(PAGE_SIZE, GFP_KERNEL); if (!conf->stripe_hashtbl) goto abort;
/* We init hash_locks[0] separately to that it can be used *asthereferencelockinthespin_lock_nest_lock()call *inlock_all_device_hash_locks_irqinordertoconvince *lockdepthatweknowwhatwearedoing.
*/
spin_lock_init(conf->hash_locks); for (i = 1; i < NR_STRIPE_HASH_LOCKS; i++)
spin_lock_init(conf->hash_locks + i);
for (i = 0; i < NR_STRIPE_HASH_LOCKS; i++)
INIT_LIST_HEAD(conf->inactive_list + i);
for (i = 0; i < NR_STRIPE_HASH_LOCKS; i++)
INIT_LIST_HEAD(conf->temp_inactive_list + i);
abort: if (conf)
free_conf(conf); return ERR_PTR(ret);
}
staticint only_parity(int raid_disk, int algo, int raid_disks, int max_degraded)
{ switch (algo) { case ALGORITHM_PARITY_0: if (raid_disk < max_degraded) return1; break; case ALGORITHM_PARITY_N: if (raid_disk >= raid_disks - max_degraded) return1; break; case ALGORITHM_PARITY_0_6: if (raid_disk == 0 ||
raid_disk == raid_disks - 1) return1; break; case ALGORITHM_LEFT_ASYMMETRIC_6: case ALGORITHM_RIGHT_ASYMMETRIC_6: case ALGORITHM_LEFT_SYMMETRIC_6: case ALGORITHM_RIGHT_SYMMETRIC_6: if (raid_disk == raid_disks - 1) return1;
} return0;
}
if ((test_bit(MD_HAS_JOURNAL, &mddev->flags) || journal_dev) &&
(mddev->bitmap_info.offset || mddev->bitmap_info.file)) {
pr_notice("md/raid:%s: array cannot have both journal and bitmap\n",
mdname(mddev)); return -EINVAL;
}
if (mddev->reshape_position != MaxSector) { /* Check that we can continue the reshape. *Difficultiesariseifthestripewewouldwriteto *nextisatorafterthestripewewouldreadfromnext. *Forareshapethatchangesthenumberofdevices,this *isonlypossibleforaveryshorttime,andmdadmmakes *surethattimeappearstohavepastbeforeassembling *thearray.Sowefailifthattimehasn'tpassed. *Forareshapethatkeepsthenumberofdevicesthesame *mdadmmustbemonitoringthereshapecankeepingthe *criticalareasread-onlyandbackedup.Itwillstart *thearrayinread-onlymode,sowecheckforthat.
*/
sector_t here_new, here_old; int old_disks; int max_degraded = (mddev->level == 6 ? 2 : 1); int chunk_sectors; int new_data_disks;
if (journal_dev) {
pr_warn("md/raid:%s: don't support reshape with journal - aborting.\n",
mdname(mddev)); return -EINVAL;
}
if (mddev->new_level != mddev->level) {
pr_warn("md/raid:%s: unsupported reshape required - aborting.\n",
mdname(mddev)); return -EINVAL;
}
old_disks = mddev->raid_disks - mddev->delta_disks; /* reshape_position must be on a new-stripe boundary, and one *furtherupinnewgeometrymustmapafterhereinold *geometry. *Ifthechunksizesaredifferent,thenasweperformreshape *inunitsofthelargestofthetwo,reshape_positionneeds *beamultipleofthelargestchunksizetimesnewdatadisks.
*/
here_new = mddev->reshape_position;
chunk_sectors = max(mddev->chunk_sectors, mddev->new_chunk_sectors);
new_data_disks = mddev->raid_disks - max_degraded; if (sector_div(here_new, chunk_sectors * new_data_disks)) {
pr_warn("md/raid:%s: reshape_position not on a stripe boundary\n",
mdname(mddev)); return -EINVAL;
}
reshape_offset = here_new * chunk_sectors; /* here_new is the stripe we will write to */
here_old = mddev->reshape_position;
sector_div(here_old, chunk_sectors * (old_disks-max_degraded)); /* here_old is the first stripe that we might need to read
* from */ if (mddev->delta_disks == 0) { /* We cannot be sure it is safe to start an in-place *reshape.Itisonlysafeifuser-spaceismonitoring *andtakingconstantbackups. *mdadmalwaysstartsasituationlikethisin *readonlymodesoitcantakecontrolbefore *allowinganywrites.Sojustcheckforthat.
*/ if (abs(min_offset_diff) >= mddev->chunk_sectors &&
abs(min_offset_diff) >= mddev->new_chunk_sectors) /* not really in-place - so OK */; elseif (mddev->ro == 0) {
pr_warn("md/raid:%s: in-place reshape must be started in read-only mode - aborting\n",
mdname(mddev)); return -EINVAL;
}
} elseif (mddev->reshape_backwards
? (here_new * chunk_sectors + min_offset_diff <=
here_old * chunk_sectors)
: (here_new * chunk_sectors >=
here_old * chunk_sectors + (-min_offset_diff))) { /* Reading from the same stripe as writing to - bad */
pr_warn("md/raid:%s: reshape_position too early for auto-recovery - aborting.\n",
mdname(mddev)); return -EINVAL;
}
pr_debug("md/raid:%s: reshape will continue\n", mdname(mddev)); /* OK, we should be able to continue; */
} else {
BUG_ON(mddev->level != mddev->new_level);
BUG_ON(mddev->layout != mddev->new_layout);
BUG_ON(mddev->chunk_sectors != mddev->new_chunk_sectors);
BUG_ON(mddev->delta_disks != 0);
}
if (test_bit(MD_HAS_JOURNAL, &mddev->flags) &&
test_bit(MD_HAS_PPL, &mddev->flags)) {
pr_warn("md/raid:%s: using journal device and PPL not allowed - disabling PPL\n",
mdname(mddev));
clear_bit(MD_HAS_PPL, &mddev->flags);
clear_bit(MD_HAS_MULTIPLE_PPLS, &mddev->flags);
}
for (i = 0; i < conf->raid_disks && conf->previous_raid_disks;
i++) {
rdev = conf->disks[i].rdev; if (!rdev) continue; if (conf->disks[i].replacement &&
conf->reshape_progress != MaxSector) { /* replacements and reshape simply do not mix. */
pr_warn("md: cannot handle concurrent replacement and reshape.\n"); goto abort;
} if (test_bit(In_sync, &rdev->flags)) continue; /* This disc is not fully in-sync. However if it *juststoredparity(beyondtherecovery_offset), *whenwedon'tneedtobeconcernedaboutthe *arraybeingdirty. *Whenreshapegoes'backwards',weneverhave *partiallycompleteddevices,soweonlyneed *toworryaboutreshapegoingforwards.
*/ /* Hack because v0.91 doesn't store recovery_offset properly. */ if (mddev->major_version == 0 &&
mddev->minor_version > 90)
rdev->recovery_offset = reshape_offset;
if (rdev->recovery_offset < reshape_offset) { /* We need to check old and new layout */ if (!only_parity(rdev->raid_disk,
conf->algorithm,
conf->raid_disks,
conf->max_degraded)) continue;
} if (!only_parity(rdev->raid_disk,
conf->prev_algo,
conf->previous_raid_disks,
conf->max_degraded)) continue;
dirty_parity_disks++;
}
if (has_failed(conf)) {
pr_crit("md/raid:%s: not enough operational devices (%d/%d failed)\n",
mdname(mddev), mddev->degraded, conf->raid_disks); goto abort;
}
/* device size must be a multiple of chunk size */
mddev->dev_sectors &= ~((sector_t)mddev->chunk_sectors - 1);
mddev->resync_max_sectors = mddev->dev_sectors;
pr_info("md/raid:%s: raid level %d active with %d out of %d devices, algorithm %d\n",
mdname(mddev), conf->level,
mddev->raid_disks-mddev->degraded, mddev->raid_disks,
mddev->new_layout);
/* Ok, everything is just fine now */ if (mddev->to_remove == &raid5_attrs_group)
mddev->to_remove = NULL; elseif (mddev->kobj.sd &&
sysfs_create_group(&mddev->kobj, &raid5_attrs_group))
pr_warn("raid5: failed to create sysfs attributes for %s\n",
mdname(mddev));
md_set_array_sectors(mddev, raid5_size(mddev, 0, 0));
if (!mddev_is_dm(mddev)) {
ret = raid5_set_limits(mddev); if (ret) goto abort;
}
if (log_init(conf, journal_dev, raid5_has_ppl(conf))) goto abort;
return0;
abort:
md_unregister_thread(mddev, &mddev->thread);
print_raid5_conf(conf);
free_conf(conf);
mddev->private = NULL;
pr_warn("md/raid:%s: failed to run raid set.\n", mdname(mddev)); return ret;
}
for (i = 0; i < conf->raid_disks; i++) {
rdev = conf->disks[i].rdev; if (rdev)
pr_debug(" disk %d, o:%d, dev:%pg\n",
i, !test_bit(Faulty, &rdev->flags),
rdev->bdev);
}
}
if (number >= conf->raid_disks &&
conf->reshape_progress == MaxSector)
clear_bit(In_sync, &rdev->flags);
if (test_bit(In_sync, &rdev->flags) ||
atomic_read(&rdev->nr_pending)) {
err = -EBUSY; goto abort;
} /* Only remove non-faulty devices if recovery *isn'tpossible.
*/ if (!test_bit(Faulty, &rdev->flags) &&
mddev->recovery_disabled != conf->recovery_disabled &&
!has_failed(conf) &&
(!p->replacement || p->replacement == rdev) &&
number < conf->raid_disks) {
err = -EBUSY; goto abort;
}
WRITE_ONCE(*rdevp, NULL); if (!err) {
err = log_modify(conf, rdev, false); if (err) goto abort;
}
tmp = p->replacement; if (tmp) { /* We must have just cleared 'rdev' */
WRITE_ONCE(p->rdev, tmp);
clear_bit(Replacement, &tmp->flags);
WRITE_ONCE(p->replacement, NULL);
if (!err)
err = log_modify(conf, tmp, true);
}
clear_bit(WantReplacement, &rdev->flags);
abort:
print_raid5_conf(conf); return err;
}
staticint raid5_add_disk(struct mddev *mddev, struct md_rdev *rdev)
{ struct r5conf *conf = mddev->private; int ret, err = -EEXIST; int disk; struct disk_info *p; struct md_rdev *tmp; int first = 0; int last = conf->raid_disks - 1;
if (test_bit(Journal, &rdev->flags)) { if (conf->log) return -EBUSY;
rdev->raid_disk = 0; /* *Thearrayisinreadonlymodeifjournalismissing,sono *writerequestsrunning.Weshouldbesafe
*/
ret = log_init(conf, rdev, false); if (ret) return ret;
ret = r5l_start(conf->log); if (ret) return ret;
return0;
} if (mddev->recovery_disabled == conf->recovery_disabled) return -EBUSY;
if (rdev->saved_raid_disk < 0 && has_failed(conf)) /* no point adding a device */ return -EINVAL;
if (rdev->raid_disk >= 0)
first = last = rdev->raid_disk;
/* *findthedisk...butpreferrdev->saved_raid_disk *ifpossible.
*/ if (rdev->saved_raid_disk >= first &&
rdev->saved_raid_disk <= last &&
conf->disks[rdev->saved_raid_disk].rdev == NULL)
first = rdev->saved_raid_disk;
for (disk = first; disk <= last; disk++) {
p = conf->disks + disk; if (p->rdev == NULL) {
clear_bit(In_sync, &rdev->flags);
rdev->raid_disk = disk; if (rdev->saved_raid_disk != disk)
conf->fullsync = 1;
WRITE_ONCE(p->rdev, rdev);
staticint raid5_resize(struct mddev *mddev, sector_t sectors)
{ /* no resync is happening, and there is enough space *onalldevices,sowecanresize. *Weneedtomakesureresynccoversanynewspace. *Ifthearrayisshrinkingweshouldpossiblywaituntil *anyiointheremovedspacecompletes,butithardlyseems *worthit.
*/
sector_t newsize; struct r5conf *conf = mddev->private; int ret;
if (raid5_has_log(conf) || raid5_has_ppl(conf)) return -EINVAL; if (mddev->delta_disks == 0 &&
mddev->new_layout == mddev->layout &&
mddev->new_chunk_sectors == mddev->chunk_sectors) return0; /* nothing to do */ if (has_failed(conf)) return -EINVAL; if (mddev->delta_disks < 0 && mddev->reshape_position == MaxSector) { /* We might be able to shrink, but the devices must *bemadebiggerfirst. *Forraid6,4istheminimumsize. *Otherwise2istheminimum
*/ int min = 2; if (mddev->level == 6)
min = 4; if (mddev->raid_disks + mddev->delta_disks < min) return -EINVAL;
}
if (test_bit(MD_RECOVERY_RUNNING, &mddev->recovery)) return -EBUSY;
if (!check_stripe_cache(mddev)) return -ENOSPC;
if (has_failed(conf)) return -EINVAL;
/* raid5 can't handle concurrent reshape and recovery */ if (mddev->resync_offset < MaxSector) return -EBUSY; for (i = 0; i < conf->raid_disks; i++) if (conf->disks[i].replacement) return -EBUSY;
if (spares - mddev->degraded < mddev->delta_disks - conf->max_degraded) /* Not enough devices even to make a degraded array *ofthatsize
*/ return -EINVAL;
/* Refuse to reduce size of the array. Any reductions in *arraysizemustbethroughexplicitsettingofarray_size *attribute.
*/ if (raid5_size(mddev, 0, conf->raid_disks + mddev->delta_disks)
< mddev->array_sectors) {
pr_warn("md/raid:%s: array size must be reduced before number of disks\n",
mdname(mddev)); return -EINVAL;
}
/* Now make sure any requests that proceeded on the assumption *thereshapewasn'trunning-likeDiscardorRead-have *completed.
*/
raid5_quiesce(mddev, true);
raid5_quiesce(mddev, false);
/* Add some new drives, as many as will fit. *Weknowthereareenoughtomakethenewlysizedarraywork. *Don'tadddevicesifwearereducingthenumberof *devicesinthearray.Thisisbecauseitisnotpossible *tocorrectlyrecordthe"partiallyreconstructed"stateof *suchdevicesduringthereshapeandconfusioncouldresult.
*/ if (mddev->delta_disks >= 0) {
rdev_for_each(rdev, mddev) if (rdev->raid_disk < 0 &&
!test_bit(Faulty, &rdev->flags)) { if (raid5_add_disk(mddev, rdev) == 0) { if (rdev->raid_disk
>= conf->previous_raid_disks)
set_bit(In_sync, &rdev->flags); else
rdev->recovery_offset = 0;
/* Failure here is OK */
sysfs_link_rdev(mddev, rdev);
}
} elseif (rdev->raid_disk >= conf->previous_raid_disks
&& !test_bit(Faulty, &rdev->flags)) { /* This is a spare that was manually added */
set_bit(In_sync, &rdev->flags);
}
/* When a reshape changes the number of devices, *->degradedismeasuredagainstthelargerofthe *preandpostnumberofdevices.
*/
spin_lock_irqsave(&conf->device_lock, flags);
mddev->degraded = raid5_calc_degraded(conf);
spin_unlock_irqrestore(&conf->device_lock, flags);
}
mddev->raid_disks = conf->raid_disks;
mddev->reshape_position = conf->reshape_progress;
set_bit(MD_SB_CHANGE_DEVS, &mddev->sb_flags);
/* This is called from the raid5d thread with mddev_lock held. *Itmakesconfigchangestothedevice.
*/ staticvoid raid5_finish_reshape(struct mddev *mddev)
{ struct r5conf *conf = mddev->private; struct md_rdev *rdev;
if (!test_bit(MD_RECOVERY_INTR, &mddev->recovery)) {
if (mddev->delta_disks <= 0) { int d;
spin_lock_irq(&conf->device_lock);
mddev->degraded = raid5_calc_degraded(conf);
spin_unlock_irq(&conf->device_lock); for (d = conf->raid_disks ;
d < conf->raid_disks - mddev->delta_disks;
d++) {
rdev = conf->disks[d].rdev; if (rdev)
clear_bit(In_sync, &rdev->flags);
rdev = conf->disks[d].replacement; if (rdev)
clear_bit(In_sync, &rdev->flags);
}
}
mddev->layout = conf->algorithm;
mddev->chunk_sectors = conf->chunk_sectors;
mddev->reshape_position = MaxSector;
mddev->delta_disks = 0;
mddev->reshape_backwards = 0;
}
}
/* for raid0 takeover only one zone is supported */ if (raid0_conf->nr_strip_zones > 1) {
pr_warn("md/raid:%s: cannot takeover raid0 with more than one zone.\n",
mdname(mddev)); return ERR_PTR(-EINVAL);
}
sectors = raid0_conf->strip_zone[0].zone_end;
sector_div(sectors, raid0_conf->strip_zone[0].nb_dev);
mddev->dev_sectors = sectors;
mddev->new_level = level;
mddev->new_layout = ALGORITHM_PARITY_N;
mddev->new_chunk_sectors = mddev->chunk_sectors;
mddev->raid_disks += 1;
mddev->delta_disks = 1; /* make sure it will be not marked as dirty */
mddev->resync_offset = MaxSector;
return setup_conf(mddev);
}
staticvoid *raid5_takeover_raid1(struct mddev *mddev)
{ int chunksect; void *ret;
if (mddev->raid_disks != 2 ||
mddev->degraded > 1) return ERR_PTR(-EINVAL);
/* Should check if there are write-behind devices? */
chunksect = 64*2; /* 64K by default */
/* The array must be an exact multiple of chunksize */ while (chunksect && (mddev->array_sectors & (chunksect-1)))
chunksect >>= 1;
if ((chunksect<<9) < RAID5_STRIPE_SIZE((struct r5conf *)mddev->private)) /* array size does not allow a suitable chunk size */ return ERR_PTR(-EINVAL);
ret = setup_conf(mddev); if (!IS_ERR(ret))
mddev_clear_unsupported_flags(mddev,
UNSUPPORTED_MDDEV_FLAGS); return ret;
}
staticvoid *raid5_takeover_raid6(struct mddev *mddev)
{ int new_layout;
switch (mddev->layout) { case ALGORITHM_LEFT_ASYMMETRIC_6:
new_layout = ALGORITHM_LEFT_ASYMMETRIC; break; case ALGORITHM_RIGHT_ASYMMETRIC_6:
new_layout = ALGORITHM_RIGHT_ASYMMETRIC; break; case ALGORITHM_LEFT_SYMMETRIC_6:
new_layout = ALGORITHM_LEFT_SYMMETRIC; break; case ALGORITHM_RIGHT_SYMMETRIC_6:
new_layout = ALGORITHM_RIGHT_SYMMETRIC; break; case ALGORITHM_PARITY_0_6:
new_layout = ALGORITHM_PARITY_0; break; case ALGORITHM_PARITY_N:
new_layout = ALGORITHM_PARITY_N; break; default: return ERR_PTR(-EINVAL);
}
mddev->new_level = 5;
mddev->new_layout = new_layout;
mddev->delta_disks = -1;
mddev->raid_disks -= 1; return setup_conf(mddev);
}
staticint raid5_check_reshape(struct mddev *mddev)
{ /* For a 2-drive array, the layout and chunk size can be changed *immediatelyasnotrestripingisneeded. *Forlargerarrayswerecordthenewvalue-aftervalidation *tobeusedbyareshapepass.
*/ struct r5conf *conf = mddev->private; int new_chunk = mddev->new_chunk_sectors;
if (mddev->new_layout >= 0 && !algorithm_valid_raid5(mddev->new_layout)) return -EINVAL; if (new_chunk > 0) { if (!is_power_of_2(new_chunk)) return -EINVAL; if (new_chunk < (PAGE_SIZE>>9)) return -EINVAL; if (mddev->array_sectors & (new_chunk-1)) /* not factor of array size */ return -EINVAL;
}
/* They look valid */
if (mddev->raid_disks == 2) { /* can make the change immediately */ if (mddev->new_layout >= 0) {
conf->algorithm = mddev->new_layout;
mddev->layout = mddev->new_layout;
} if (new_chunk > 0) {
conf->chunk_sectors = new_chunk ;
mddev->chunk_sectors = new_chunk;
}
set_bit(MD_SB_CHANGE_DEVS, &mddev->sb_flags);
md_wakeup_thread(mddev->thread);
} return check_reshape(mddev);
}
staticint raid6_check_reshape(struct mddev *mddev)
{ int new_chunk = mddev->new_chunk_sectors;
if (mddev->new_layout >= 0 && !algorithm_valid_raid6(mddev->new_layout)) return -EINVAL; if (new_chunk > 0) { if (!is_power_of_2(new_chunk)) return -EINVAL; if (new_chunk < (PAGE_SIZE >> 9)) return -EINVAL; if (mddev->array_sectors & (new_chunk-1)) /* not factor of array size */ return -EINVAL;
}
/* They look valid */ return check_reshape(mddev);
}
staticvoid *raid5_takeover(struct mddev *mddev)
{ /* raid5 can take over: *raid0-ifthereisonlyonestripzone-makeitaraid4layout *raid1-iftherearetwodrives.Weneedtoknowthechunksize *raid4-trivial-justusearaid4layout. *raid6-Providingitisa*_6layout
*/ if (mddev->level == 0) return raid45_takeover_raid0(mddev, 5); if (mddev->level == 1) return raid5_takeover_raid1(mddev); if (mddev->level == 4) {
mddev->new_layout = ALGORITHM_PARITY_N;
mddev->new_level = 5; return setup_conf(mddev);
} if (mddev->level == 6) return raid5_takeover_raid6(mddev);
staticvoid *raid6_takeover(struct mddev *mddev)
{ /* Currently can only take over a raid5. We map the *personalitytoanequivalentraid6personality *withtheQblockattheend.
*/ int new_layout;
if (mddev->pers != &raid5_personality) return ERR_PTR(-EINVAL); if (mddev->degraded > 1) return ERR_PTR(-EINVAL); if (mddev->raid_disks > 253) return ERR_PTR(-EINVAL); if (mddev->raid_disks < 3) return ERR_PTR(-EINVAL);
switch (mddev->layout) { case ALGORITHM_LEFT_ASYMMETRIC:
new_layout = ALGORITHM_LEFT_ASYMMETRIC_6; break; case ALGORITHM_RIGHT_ASYMMETRIC:
new_layout = ALGORITHM_RIGHT_ASYMMETRIC_6; break; case ALGORITHM_LEFT_SYMMETRIC:
new_layout = ALGORITHM_LEFT_SYMMETRIC_6; break; case ALGORITHM_RIGHT_SYMMETRIC:
new_layout = ALGORITHM_RIGHT_SYMMETRIC_6; break; case ALGORITHM_PARITY_0:
new_layout = ALGORITHM_PARITY_0_6; break; case ALGORITHM_PARITY_N:
new_layout = ALGORITHM_PARITY_N; break; default: return ERR_PTR(-EINVAL);
}
mddev->new_level = 6;
mddev->new_layout = new_layout;
mddev->delta_disks = 1;
mddev->raid_disks += 1; return setup_conf(mddev);
}
/* This used to be two separate modules, they were: */
MODULE_ALIAS("raid5");
MODULE_ALIAS("raid6");
Messung V0.5 in Prozent
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.669Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-09-30)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.