/* This aids in debugging situations where a bad LVB might be involved. */ staticvoid ocfs2_dump_meta_lvb_info(u64 level, constchar *function, unsignedint line, struct ocfs2_lock_res *lockres)
{ struct ocfs2_meta_lvb *lvb = ocfs2_dlm_lvb(&lockres->l_lksb);
void ocfs2_lock_res_init_once(struct ocfs2_lock_res *res)
{ /* This also clears out the lock status block */
memset(res, 0, sizeof(struct ocfs2_lock_res));
spin_lock_init(&res->l_lock);
init_waitqueue_head(&res->l_event);
INIT_LIST_HEAD(&res->l_blocked_list);
INIT_LIST_HEAD(&res->l_mask_waiters);
INIT_LIST_HEAD(&res->l_holders);
}
staticvoid ocfs2_super_lock_res_init(struct ocfs2_lock_res *res, struct ocfs2_super *osb)
{ /* Superblock lockres doesn't come from a slab so we call init
* once on it manually. */
ocfs2_lock_res_init_once(res);
ocfs2_build_lock_name(OCFS2_LOCK_TYPE_SUPER, OCFS2_SUPER_BLOCK_BLKNO, 0, res->l_name);
ocfs2_lock_res_init_common(osb, res, OCFS2_LOCK_TYPE_SUPER,
&ocfs2_super_lops, osb);
}
staticvoid ocfs2_rename_lock_res_init(struct ocfs2_lock_res *res, struct ocfs2_super *osb)
{ /* Rename lockres doesn't come from a slab so we call init
* once on it manually. */
ocfs2_lock_res_init_once(res);
ocfs2_build_lock_name(OCFS2_LOCK_TYPE_RENAME, 0, 0, res->l_name);
ocfs2_lock_res_init_common(osb, res, OCFS2_LOCK_TYPE_RENAME,
&ocfs2_rename_lops, osb);
}
staticvoid ocfs2_nfs_sync_lock_res_init(struct ocfs2_lock_res *res, struct ocfs2_super *osb)
{ /* nfs_sync lockres doesn't come from a slab so we call init
* once on it manually. */
ocfs2_lock_res_init_once(res);
ocfs2_build_lock_name(OCFS2_LOCK_TYPE_NFS_SYNC, 0, 0, res->l_name);
ocfs2_lock_res_init_common(osb, res, OCFS2_LOCK_TYPE_NFS_SYNC,
&ocfs2_nfs_sync_lops, osb);
}
staticinlinevoid ocfs2_inc_holders(struct ocfs2_lock_res *lockres, int level)
{
BUG_ON(!lockres);
switch(level) { case DLM_LOCK_EX:
lockres->l_ex_holders++; break; case DLM_LOCK_PR:
lockres->l_ro_holders++; break; default:
BUG();
}
}
staticinlinevoid ocfs2_dec_holders(struct ocfs2_lock_res *lockres, int level)
{
BUG_ON(!lockres);
switch(level) { case DLM_LOCK_EX:
BUG_ON(!lockres->l_ex_holders);
lockres->l_ex_holders--; break; case DLM_LOCK_PR:
BUG_ON(!lockres->l_ro_holders);
lockres->l_ro_holders--; break; default:
BUG();
}
}
/* WARNING: This function lives in a world where the only three lock *levelsareEX,PR,andNL.It*will*havetobeadjustedwhenmore
* lock types are added. */ staticinlineint ocfs2_highest_compat_lock_level(int level)
{ int new_level = DLM_LOCK_EX;
/* Convert from RO to EX doesn't really need anything as our *informationisalreadyuptodata.ConvertfromNLto **anything*howevershouldmarkourselvesasneedingan
* update */ if (lockres->l_level == DLM_LOCK_NL &&
lockres->l_ops->flags & LOCK_TYPE_REQUIRES_REFRESH)
lockres_or_flags(lockres, OCFS2_LOCK_NEEDS_REFRESH);
staticint ocfs2_generic_handle_bast(struct ocfs2_lock_res *lockres, int level)
{ int needs_downconvert = 0;
assert_spin_locked(&lockres->l_lock);
if (level > lockres->l_blocking) { /* only schedule a downconvert if we haven't already scheduled *onethatgoeslowenoughtosatisfythelevelwe're *blocking.thisalsocatchesthecasewhereweget
* duplicate BASTs */ if (ocfs2_highest_compat_lock_level(level) <
ocfs2_highest_compat_lock_level(lockres->l_blocking))
needs_downconvert = 1;
if (status == -EAGAIN) {
lockres_clear_flags(lockres, OCFS2_LOCK_BUSY); goto out;
}
if (status) {
mlog(ML_ERROR, "lockres %s: lksb status value of %d!\n",
lockres->l_name, status);
spin_unlock_irqrestore(&lockres->l_lock, flags); return;
}
switch(lockres->l_action) { case OCFS2_AST_ATTACH:
ocfs2_generic_handle_attach_action(lockres);
lockres_clear_flags(lockres, OCFS2_LOCK_LOCAL); break; case OCFS2_AST_CONVERT:
ocfs2_generic_handle_convert_action(lockres); break; case OCFS2_AST_DOWNCONVERT:
ocfs2_generic_handle_downconvert_action(lockres); break; default:
mlog(ML_ERROR, "lockres %s: AST fired with invalid action: %u, " "flags 0x%lx, unlock: %u\n",
lockres->l_name, lockres->l_action, lockres->l_flags,
lockres->l_unlock_action);
BUG();
}
out: /* set it to something invalid so if we get called again we
* can catch it. */
lockres->l_action = OCFS2_AST_INVALID;
/* Did we try to cancel this lock? Clear that state */ if (lockres->l_unlock_action == OCFS2_UNLOCK_CANCEL_CONVERT)
lockres->l_unlock_action = OCFS2_UNLOCK_INVALID;
switch(lockres->l_unlock_action) { case OCFS2_UNLOCK_CANCEL_CONVERT:
mlog(0, "Cancel convert success for %s\n", lockres->l_name);
lockres->l_action = OCFS2_AST_INVALID; /* Downconvert thread may have requeued this lock, we
* need to wake it. */ if (lockres->l_flags & OCFS2_LOCK_BLOCKED)
ocfs2_wake_downconvert_thread(ocfs2_get_lockres_osb(lockres)); break; case OCFS2_UNLOCK_DROP_LOCK:
lockres->l_level = DLM_LOCK_IV; break; default:
BUG();
}
/* Note: If we detect another process working on the lock (i.e., *OCFS2_LOCK_BUSY),we'llbailoutreturning0.It'suptothecaller *todotherightthinginthatcase.
*/ staticint ocfs2_lock_create(struct ocfs2_super *osb, struct ocfs2_lock_res *lockres, int level,
u32 dlm_flags)
{ int ret = 0; unsignedlong flags; unsignedint gen;
/* predict what lock level we'll be dropping down to on behalf *ofanothernode,andreturntrueifthecurrentlywanted
* level will be compatible with it. */ staticinlineint ocfs2_may_continue_on_blocked_lock(struct ocfs2_lock_res *lockres, int wanted)
{
BUG_ON(!(lockres->l_flags & OCFS2_LOCK_BLOCKED));
staticint ocfs2_wait_for_mask(struct ocfs2_mask_waiter *mw)
{
wait_for_completion(&mw->mw_complete); /* Re-arm the completion in case we want to wait on it again */
reinit_completion(&mw->mw_complete); return mw->mw_status;
}
/* returns 0 if the mw that was removed was already satisfied, -EBUSY
* if the mask still hadn't reached its goal */ staticint __lockres_remove_mask_waiter(struct ocfs2_lock_res *lockres, struct ocfs2_mask_waiter *mw)
{ int ret = 0;
assert_spin_locked(&lockres->l_lock); if (!list_empty(&mw->mw_item)) { if ((lockres->l_flags & mw->mw_mask) != mw->mw_goal)
ret = -EBUSY;
staticint lockres_remove_mask_waiter(struct ocfs2_lock_res *lockres, struct ocfs2_mask_waiter *mw)
{ unsignedlong flags; int ret = 0;
spin_lock_irqsave(&lockres->l_lock, flags);
ret = __lockres_remove_mask_waiter(lockres, mw);
spin_unlock_irqrestore(&lockres->l_lock, flags);
return ret;
}
staticint ocfs2_wait_for_mask_interruptible(struct ocfs2_mask_waiter *mw, struct ocfs2_lock_res *lockres)
{ int ret;
ret = wait_for_completion_interruptible(&mw->mw_complete); if (ret)
lockres_remove_mask_waiter(lockres, mw); else
ret = mw->mw_status; /* Re-arm the completion in case we want to wait on it again */
reinit_completion(&mw->mw_complete); return ret;
}
staticint __ocfs2_cluster_lock(struct ocfs2_super *osb, struct ocfs2_lock_res *lockres, int level,
u32 lkm_flags, int arg_flags, int l_subclass, unsignedlong caller_ip)
{ struct ocfs2_mask_waiter mw; int wait, catch_signals = !(osb->s_mount_opt & OCFS2_MOUNT_NOINTR); int ret = 0; /* gcc doesn't realize wait = 1 guarantees ret is set */ unsignedlong flags; unsignedint gen; int noqueue_attempted = 0; int dlm_locked = 0; int kick_dc = 0;
if (!(lockres->l_flags & OCFS2_LOCK_INITIALIZED)) {
mlog_errno(-EINVAL); return -EINVAL;
}
ocfs2_init_mask_waiter(&mw);
if (lockres->l_ops->flags & LOCK_TYPE_USES_LVB)
lkm_flags |= DLM_LKF_VALBLK;
again:
wait = 0;
spin_lock_irqsave(&lockres->l_lock, flags);
if (catch_signals && signal_pending(current)) {
ret = -ERESTARTSYS; goto unlock;
}
mlog_bug_on_msg(lockres->l_flags & OCFS2_LOCK_FREEING, "Cluster lock called on freeing lockres %s! flags " "0x%lx\n", lockres->l_name, lockres->l_flags);
/* We only compare against the currently granted level *here.Ifthelockisblockedwaitingonadownconvert,
* we'll get caught below. */ if (lockres->l_flags & OCFS2_LOCK_BUSY &&
level > lockres->l_level) { /* is someone sitting in dlm_lock? If so, wait on
* them. */
lockres_add_mask_waiter(lockres, &mw, OCFS2_LOCK_BUSY, 0);
wait = 1; goto unlock;
}
if (lockres->l_flags & OCFS2_LOCK_BLOCKED &&
!ocfs2_may_continue_on_blocked_lock(lockres, level)) { /* is the lock is currently blocked on behalf of
* another node */
lockres_add_mask_waiter(lockres, &mw, OCFS2_LOCK_BLOCKED, 0);
wait = 1; goto unlock;
}
if (level > lockres->l_level) { if (noqueue_attempted > 0) {
ret = -EAGAIN; goto unlock;
} if (lkm_flags & DLM_LKF_NOQUEUE)
noqueue_attempted = 1;
if (lockres->l_action != OCFS2_AST_INVALID)
mlog(ML_ERROR, "lockres %s has action %u pending\n",
lockres->l_name, lockres->l_action);
/* Grants us an EX lock on the data and metadata resources, skipping *thenormalclusterdirectorylookup.UsethisONLYonnewlycreated *inodeswhichothernodescan'tpossiblysee,andwhichhaven'tbeen *hashedintheinodehashyet.Thiscangiveusagoodperformance *increaseasit'llskipthenetworkbroadcastnormallyassociated
* with creating a new lock resource. */ int ocfs2_create_new_inode_locks(struct inode *inode)
{ int ret; struct ocfs2_super *osb = OCFS2_SB(inode->i_sb);
/* NOTE: That we don't increment any of the holder counts, nor *doweaddanythingtoajournalhandle.Sincethisis *supposedtobeanewinodewhichtheclusterdoesn'tknow *aboutyet,thereisnoneedto.AsfarastheLVBhandling *isconcerned,thisisbasicallylikeacquiringanEXlock *onaresourcewhichhasaninvalidone--we'llsetit
* valid when we release the EX. */
ret = ocfs2_create_new_lock(osb, &OCFS2_I(inode)->ip_rw_lockres, 1, 1); if (ret) {
mlog_errno(ret); goto bail;
}
/* *Wedon'twanttouseDLM_LKF_LOCALonametadatalockasthey *don'tuseagenerationintheirlocknames.
*/
ret = ocfs2_create_new_lock(osb, &OCFS2_I(inode)->ip_inode_lockres, 1, 0); if (ret) {
mlog_errno(ret); goto bail;
}
ret = ocfs2_create_new_lock(osb, &OCFS2_I(inode)->ip_open_lockres, 0, 0); if (ret)
mlog_errno(ret);
bail: return ret;
}
int ocfs2_rw_lock(struct inode *inode, int write)
{ int status, level; struct ocfs2_lock_res *lockres; struct ocfs2_super *osb = OCFS2_SB(inode->i_sb);
if (!ocfs2_mount_local(osb))
ocfs2_cluster_unlock(osb, lockres, level);
}
/* *ocfs2_open_lockalwaysgetPRmodelock.
*/ int ocfs2_open_lock(struct inode *inode)
{ int status = 0; struct ocfs2_lock_res *lockres; struct ocfs2_super *osb = OCFS2_SB(inode->i_sb);
mlog(0, "inode %llu take PRMODE open lock\n",
(unsignedlonglong)OCFS2_I(inode)->ip_blkno);
if (ocfs2_is_hard_readonly(osb) || ocfs2_mount_local(osb)) goto out;
lockres = &OCFS2_I(inode)->ip_open_lockres;
status = ocfs2_cluster_lock(osb, lockres, DLM_LOCK_PR, 0, 0); if (status < 0)
mlog_errno(status);
out: return status;
}
int ocfs2_try_open_lock(struct inode *inode, int write)
{ int status = 0, level; struct ocfs2_lock_res *lockres; struct ocfs2_super *osb = OCFS2_SB(inode->i_sb);
mlog(0, "inode %llu try to take %s open lock\n",
(unsignedlonglong)OCFS2_I(inode)->ip_blkno,
write ? "EXMODE" : "PRMODE");
if (ocfs2_is_hard_readonly(osb)) { if (write)
status = -EROFS; goto out;
}
/* If we know that another node is waiting on our lock, kick *thedownconvertthread*pre-emptivelywhenwereacharelease
* condition. */ if (lockres->l_flags & OCFS2_LOCK_BLOCKED) { switch(lockres->l_blocking) { case DLM_LOCK_EX: if (!lockres->l_ex_holders && !lockres->l_ro_holders)
kick = 1; break; case DLM_LOCK_PR: if (!lockres->l_ex_holders)
kick = 1; break; default:
BUG();
}
}
/* LVB only has room for 64 bits of time here so we pack it for
* now. */ static u64 ocfs2_pack_timespec(struct timespec64 *spec)
{
u64 res;
u64 sec = clamp_t(time64_t, spec->tv_sec, 0, 0x3ffffffffull);
u32 nsec = spec->tv_nsec;
res = (sec << OCFS2_SEC_SHIFT) | (nsec & OCFS2_NSEC_MASK);
return res;
}
/* Call this with the lockres locked. I am reasonably sure we don't *needip_lockinthisfunctionasanyonewhowouldbechangingthose
* values is supposed to be blocked in ocfs2_inode_lock right now. */ staticvoid __ocfs2_stuff_meta_lvb(struct inode *inode)
{ struct ocfs2_inode_info *oi = OCFS2_I(inode); struct ocfs2_lock_res *lockres = &oi->ip_inode_lockres; struct ocfs2_meta_lvb *lvb; struct timespec64 ts;
lvb = ocfs2_dlm_lvb(&lockres->l_lksb); if (inode_wrong_type(inode, be16_to_cpu(lvb->lvb_imode))) return -ESTALE;
/* We're safe here without the lockres lock... */
spin_lock(&oi->ip_lock);
oi->ip_clusters = be32_to_cpu(lvb->lvb_iclusters);
i_size_write(inode, be64_to_cpu(lvb->lvb_isize));
/* fast-symlinks are a special case */ if (S_ISLNK(inode->i_mode) && !oi->ip_clusters)
inode->i_blocks = 0; else
inode->i_blocks = ocfs2_inode_sector_count(inode);
/* If status is non zero, I'll mark it as not being in refresh
* anymroe, but i won't clear the needs refresh flag. */ staticinlinevoid ocfs2_complete_lock_res_refresh(struct ocfs2_lock_res *lockres, int status)
{ unsignedlong flags;
spin_lock_irqsave(&lockres->l_lock, flags);
lockres_clear_flags(lockres, OCFS2_LOCK_REFRESHING); if (!status)
lockres_clear_flags(lockres, OCFS2_LOCK_NEEDS_REFRESH);
spin_unlock_irqrestore(&lockres->l_lock, flags);
wake_up(&lockres->l_event);
}
/* may or may not return a bh if it went to disk. */ staticint ocfs2_inode_lock_update(struct inode *inode, struct buffer_head **bh)
{ int status = 0; struct ocfs2_inode_info *oi = OCFS2_I(inode); struct ocfs2_lock_res *lockres = &oi->ip_inode_lockres; struct ocfs2_dinode *fe; struct ocfs2_super *osb = OCFS2_SB(inode->i_sb);
if (ocfs2_mount_local(osb)) goto bail;
spin_lock(&oi->ip_lock); if (oi->ip_flags & OCFS2_INODE_DELETED) {
mlog(0, "Orphaned inode %llu was deleted while we " "were waiting on a lock. ip_flags = 0x%x\n",
(unsignedlonglong)oi->ip_blkno, oi->ip_flags);
spin_unlock(&oi->ip_lock);
status = -ENOENT; goto bail;
}
spin_unlock(&oi->ip_lock);
if (!ocfs2_should_refresh_lock_res(lockres)) goto bail;
/* This will discard any caching information we might have had
* for the inode metadata. */
ocfs2_metadata_cache_purge(INODE_CACHE(inode));
ocfs2_extent_map_trunc(inode, 0);
if (ocfs2_meta_lvb_is_trustable(inode, lockres)) {
mlog(0, "Trusting LVB on inode %llu\n",
(unsignedlonglong)oi->ip_blkno);
status = ocfs2_refresh_inode_from_lvb(inode); goto bail_refresh;
} else { /* Boo, we have to go to disk. */ /* read bh, cast, ocfs2_refresh_inode */
status = ocfs2_read_inode_block(inode, bh); if (status < 0) {
mlog_errno(status); goto bail_refresh;
}
fe = (struct ocfs2_dinode *) (*bh)->b_data; if (inode_wrong_type(inode, le16_to_cpu(fe->i_mode))) {
status = -ESTALE; goto bail_refresh;
}
/* This is a good chance to make sure we're not *lockinganinvalidobject.ocfs2_read_inode_block() *alreadycheckedthattheinodeblockissane. * *Webugonastaleinodeherebecausewechecked *abovewhetheritwaswipedfromdisk.Thewiping *nodeprovidesaguaranteethatwereceivethat *messageandcanmarktheinodebeforedroppingany
* locks associated with it. */
mlog_bug_on_msg(inode->i_generation !=
le32_to_cpu(fe->i_generation), "Invalid dinode %llu disk generation: %u " "inode->i_generation: %u\n",
(unsignedlonglong)oi->ip_blkno,
le32_to_cpu(fe->i_generation),
inode->i_generation);
mlog_bug_on_msg(le64_to_cpu(fe->i_dtime) ||
!(fe->i_flags & cpu_to_le32(OCFS2_VALID_FL)), "Stale dinode %llu dtime: %llu flags: 0x%x\n",
(unsignedlonglong)oi->ip_blkno,
(unsignedlonglong)le64_to_cpu(fe->i_dtime),
le32_to_cpu(fe->i_flags));
if (passed_bh) { /* Ok, the update went to disk for us, use the
* returned bh. */
*ret_bh = passed_bh;
get_bh(*ret_bh);
return0;
}
status = ocfs2_read_inode_block(inode, ret_bh); if (status < 0)
mlog_errno(status);
return status;
}
/* *returns<0errorifthecallbackwillneverbecalled,otherwise *theresultofthelockwillbecommunicatedviathecallback.
*/ int ocfs2_inode_lock_full_nested(struct inode *inode, struct buffer_head **ret_bh, int ex, int arg_flags, int subclass)
{ int status, level, acquired;
u32 dlm_flags; struct ocfs2_lock_res *lockres = NULL; struct ocfs2_super *osb = OCFS2_SB(inode->i_sb); struct buffer_head *local_bh = NULL;
mlog(0, "inode %llu, take %s META lock\n",
(unsignedlonglong)OCFS2_I(inode)->ip_blkno,
ex ? "EXMODE" : "PRMODE");
status = 0;
acquired = 0; /* We'll allow faking a readonly metadata lock for
* rodevices. */ if (ocfs2_is_hard_readonly(osb)) { if (ex)
status = -EROFS; goto getbh;
}
if ((arg_flags & OCFS2_META_LOCK_GETBH) ||
ocfs2_mount_local(osb)) goto update;
if (!(arg_flags & OCFS2_META_LOCK_RECOVERY))
ocfs2_wait_for_recovery(osb);
lockres = &OCFS2_I(inode)->ip_inode_lockres;
level = ex ? DLM_LOCK_EX : DLM_LOCK_PR;
dlm_flags = 0; if (arg_flags & OCFS2_META_LOCK_NOQUEUE)
dlm_flags |= DLM_LKF_NOQUEUE;
status = __ocfs2_cluster_lock(osb, lockres, level, dlm_flags,
arg_flags, subclass, _RET_IP_); if (status < 0) { if (status != -EAGAIN)
mlog_errno(status); goto bail;
}
/* Notify the error cleanup path to drop the cluster lock. */
acquired = 1;
/* We wait twice because a node may have died while we were in *thelowerdlmlayers.Thesecondtimethough,we've *committedtoowningthislocksowedon'tallowsignalsto
* abort the operation. */ if (!(arg_flags & OCFS2_META_LOCK_RECOVERY))
ocfs2_wait_for_recovery(osb);
update: /* *Weonlyseethisflagifwe'rebeingcalledfrom *ocfs2_read_locked_inode().Itmeanswe'relockinganinode *whichhasn'tbeenpopulatedyet,socleartherefreshflag *andletthecallerhandleit.
*/ if (inode->i_state & I_NEW) {
status = 0; if (lockres)
ocfs2_complete_lock_res_refresh(lockres, 0); goto bail;
}
/* This is fun. The caller may want a bh back, or it may *not.ocfs2_inode_lock_updatedefinitelywantsonein,but *mayormaynotreadone,dependingonwhat'sinthe *LVB.Theresultofallofthisisthatwe've*only*goneto
* disk if we have to, so the complexity is worthwhile. */
status = ocfs2_inode_lock_update(inode, &local_bh); if (status < 0) { if (status != -ENOENT)
mlog_errno(status); goto bail;
}
getbh: if (ret_bh) {
status = ocfs2_assign_bh(inode, ret_bh, local_bh); if (status < 0) {
mlog_errno(status); goto bail;
}
}
bail: if (status < 0) { if (ret_bh && (*ret_bh)) {
brelse(*ret_bh);
*ret_bh = NULL;
} if (acquired)
ocfs2_inode_unlock(inode, ex);
}
if (unlikely(ex && !tmp_oh->oh_ex)) { /* *case2.2upgrademaycausedeadlock,forbidit.
*/
mlog(ML_ERROR, "Recursive locking is not permitted to " "upgrade to EX level from PR level.\n");
dump_stack(); return -EINVAL;
}
/* *case2.1OCFS2_META_LOCK_GETBHflagmakeocfs2_inode_lock_full. *ignorethelocklevelandjustupdateit.
*/ if (ret_bh) {
status = ocfs2_inode_lock_full(inode, ret_bh, ex,
OCFS2_META_LOCK_GETBH); if (status < 0) { if (status != -ENOENT)
mlog_errno(status); return status;
}
} return1;
}
void ocfs2_inode_unlock_tracker(struct inode *inode, int ex, struct ocfs2_lock_holder *oh, int had_lock)
{ struct ocfs2_lock_res *lockres;
lockres = &OCFS2_I(inode)->ip_inode_lockres; /* had_lock means that the current process already takes the cluster *lockpreviously. *Ifhad_lockis1,wehavenothingtodohere. *Ifhad_lockis0,wewillreleasethelock.
*/ if (!had_lock) {
ocfs2_inode_unlock(inode, oh->oh_ex);
ocfs2_remove_holder(lockres, oh);
}
}
int ocfs2_orphan_scan_lock(struct ocfs2_super *osb, u32 *seqno)
{ struct ocfs2_lock_res *lockres; struct ocfs2_orphan_scan_lvb *lvb; int status = 0;
if (ocfs2_is_hard_readonly(osb)) return -EROFS;
if (ocfs2_mount_local(osb)) return0;
lockres = &osb->osb_orphan_scan.os_lockres;
status = ocfs2_cluster_lock(osb, lockres, DLM_LOCK_EX, 0, 0); if (status < 0) return status;
int ocfs2_super_lock(struct ocfs2_super *osb, int ex)
{ int status = 0; int level = ex ? DLM_LOCK_EX : DLM_LOCK_PR; struct ocfs2_lock_res *lockres = &osb->osb_super_lockres;
if (ocfs2_is_hard_readonly(osb)) return -EROFS;
if (ocfs2_mount_local(osb)) goto bail;
status = ocfs2_cluster_lock(osb, lockres, level, 0, 0); if (status < 0) {
mlog_errno(status); goto bail;
}
/* The super block lock path is really in the best position to *knowwhenresourcescoveredbythelockneedtobe *refreshed,sowedoithere.Ofcourse,makingsenseof
* everything is up to the caller :) */
status = ocfs2_should_refresh_lock_res(lockres); if (status) {
status = ocfs2_refresh_slot_info(osb);
if (!ocfs2_mount_local(osb))
ocfs2_cluster_unlock(osb, lockres,
ex ? LKM_EXMODE : LKM_PRMODE); if (ex)
up_write(&osb->nfs_sync_rwlock); else
up_read(&osb->nfs_sync_rwlock);
}
int ocfs2_trim_fs_lock(struct ocfs2_super *osb, struct ocfs2_trim_fs_info *info, int trylock)
{ int status; struct ocfs2_trim_fs_lvb *lvb; struct ocfs2_lock_res *lockres = &osb->osb_trim_fs_lockres;
if (info)
info->tf_valid = 0;
if (ocfs2_is_hard_readonly(osb)) return -EROFS;
if (ocfs2_mount_local(osb)) return0;
status = ocfs2_cluster_lock(osb, lockres, DLM_LOCK_EX,
trylock ? DLM_LKF_NOQUEUE : 0, 0); if (status < 0) { if (status != -EAGAIN)
mlog_errno(status); return status;
}
int ocfs2_dentry_lock(struct dentry *dentry, int ex)
{ int ret; int level = ex ? DLM_LOCK_EX : DLM_LOCK_PR; struct ocfs2_dentry_lock *dl = dentry->d_fsdata; struct ocfs2_super *osb = OCFS2_SB(dentry->d_sb);
BUG_ON(!dl);
if (ocfs2_is_hard_readonly(osb)) { if (ex) return -EROFS; return0;
}
if (ocfs2_mount_local(osb)) return0;
ret = ocfs2_cluster_lock(osb, &dl->dl_lockres, level, 0, 0); if (ret < 0)
mlog_errno(ret);
return ret;
}
void ocfs2_dentry_unlock(struct dentry *dentry, int ex)
{ int level = ex ? DLM_LOCK_EX : DLM_LOCK_PR; struct ocfs2_dentry_lock *dl = dentry->d_fsdata; struct ocfs2_super *osb = OCFS2_SB(dentry->d_sb);
if (!ocfs2_is_hard_readonly(osb) && !ocfs2_mount_local(osb))
ocfs2_cluster_unlock(osb, &dl->dl_lockres, level);
}
/* Reference counting of the dlm debug structure. We want this because *openreferencesonthedebuginodescanliveonafteramount,so
* we can't rely on the ocfs2_super to always exist. */ staticvoid ocfs2_dlm_debug_free(struct kref *kref)
{ struct ocfs2_dlm_debug *dlm_debug;
/* Access to this is arbitrated for us via seq_file->sem. */ struct ocfs2_dlm_seq_priv { struct ocfs2_dlm_debug *p_dlm_debug; struct ocfs2_lock_res p_iter_res; struct ocfs2_lock_res p_tmp_res;
};
list_for_each_entry(iter, &start->l_debug_list, l_debug_list) { /* discover the head of the list */ if (&iter->l_debug_list == &dlm_debug->d_lockres_tracking) {
mlog(0, "End of list found, %p\n", ret); break;
}
/* We track our "dummy" iteration lockres' by a NULL
* l_ops field. */ if (iter->l_ops != NULL) {
ret = iter; break;
}
}
spin_lock(&ocfs2_dlm_tracking_lock);
iter = ocfs2_dlm_next_res(&priv->p_iter_res, priv); if (iter) { /* Since lockres' have the lifetime of their container *(whichcanbeinodes,ocfs2_supers,etc)wewantto *copythisouttoatemporarylockreswhilestill *underthespinlock.Obviouslyafterthiswecan't *trustanypointersonthecopyreturned,butthat's *okastheinformationwewantisn'ttypicallyheld
* in them. */
priv->p_tmp_res = *iter;
iter = &priv->p_tmp_res;
}
spin_unlock(&ocfs2_dlm_tracking_lock);
/* Mark the lockres as being dropped. It will no longer be *queuedifblocking,butwestillmayhavetowaitonit *beingdequeuedfromthedownconvertthreadbeforewecanconsider *itsafetodrop. *
* You can *not* attempt to call cluster_lock on this lockres anymore. */ void ocfs2_mark_lockres_freeing(struct ocfs2_super *osb, struct ocfs2_lock_res *lockres)
{ int status; struct ocfs2_mask_waiter mw; unsignedlong flags, flags2;
ocfs2_init_mask_waiter(&mw);
spin_lock_irqsave(&lockres->l_lock, flags);
lockres->l_flags |= OCFS2_LOCK_FREEING; if (lockres->l_flags & OCFS2_LOCK_QUEUED && current == osb->dc_task) { /* *Weknowthedownconvertisqueuedbutnotinprogress *becausewearethedownconvertthreadandprocessing *differentlock.Sowecanjustremovethelockfromthe *queue.Thisisnotonlyanoptimizationbutalsoaway *toavoidthefollowingdeadlock: *ocfs2_dentry_post_unlock() *ocfs2_dentry_lock_put() *ocfs2_drop_dentry_lock() *iput() *ocfs2_evict_inode() *ocfs2_clear_inode() *ocfs2_mark_lockres_freeing() *...blockswaitingforOCFS2_LOCK_QUEUED *sincewearethedownconvertthreadwhich *shouldcleartheflag.
*/
spin_unlock_irqrestore(&lockres->l_lock, flags);
spin_lock_irqsave(&osb->dc_task_lock, flags2);
list_del_init(&lockres->l_blocked_list);
osb->blocked_lock_count--;
spin_unlock_irqrestore(&osb->dc_task_lock, flags2); /* *Warnifwerecurseintoanotherpost_unlockcall.Strictly *speakingitisn'taproblembutweneedtobecarefulif *thathappens(stackoverflow,deadlocks,...)sowarnif *ocfs2growsapathforwhichthiscanhappen.
*/
WARN_ON_ONCE(lockres->l_ops->post_unlock); /* Since the lock is freeing we don't do much in the fn below */
ocfs2_process_blocked_lock(osb, lockres); return;
} while (lockres->l_flags & OCFS2_LOCK_QUEUED) {
lockres_add_mask_waiter(lockres, &mw, OCFS2_LOCK_QUEUED, 0);
spin_unlock_irqrestore(&lockres->l_lock, flags);
mlog(0, "Waiting on lockres %s\n", lockres->l_name);
status = ocfs2_wait_for_mask(&mw); if (status)
mlog_errno(status);
/* returns 1 when the caller should unlock and call ocfs2_dlm_unlock */ staticint ocfs2_prepare_cancel_convert(struct ocfs2_super *osb, struct ocfs2_lock_res *lockres)
{
assert_spin_locked(&lockres->l_lock);
if (lockres->l_unlock_action == OCFS2_UNLOCK_CANCEL_CONVERT) { /* If we're already trying to cancel a lock conversion *thenjustdropthespinlockandallowthecallerto
* requeue this lock. */
mlog(ML_BASTS, "lockres %s, skip convert\n", lockres->l_name); return0;
}
/* were we in a convert when we got the bast fire? */
BUG_ON(lockres->l_action != OCFS2_AST_CONVERT &&
lockres->l_action != OCFS2_AST_DOWNCONVERT); /* set things up for the unlockast to know to just
* clear out the ast_action and unset busy, etc. */
lockres->l_unlock_action = OCFS2_UNLOCK_CANCEL_CONVERT;
staticint ocfs2_cancel_convert(struct ocfs2_super *osb, struct ocfs2_lock_res *lockres)
{ int ret;
ret = ocfs2_dlm_unlock(osb->cconn, &lockres->l_lksb,
DLM_LKF_CANCEL); if (ret) {
ocfs2_log_dlm_error("ocfs2_dlm_unlock", ret, lockres);
ocfs2_recover_from_dlm_error(lockres, 0);
}
mlog(ML_BASTS, "lockres %s\n", lockres->l_name);
return ret;
}
staticint ocfs2_unblock_lock(struct ocfs2_super *osb, struct ocfs2_lock_res *lockres, struct ocfs2_unblock_ctl *ctl)
{ unsignedlong flags; int blocking; int new_level; int level; int ret = 0; int set_lvb = 0; unsignedint gen;
spin_lock_irqsave(&lockres->l_lock, flags);
recheck: /* *Isitstillblocking?Ifnot,wehavenomoreworktodo.
*/ if (!(lockres->l_flags & OCFS2_LOCK_BLOCKED)) {
BUG_ON(lockres->l_blocking != DLM_LOCK_NL);
spin_unlock_irqrestore(&lockres->l_lock, flags);
ret = 0; goto leave;
}
/* if we're blocking an exclusive and we have *any* holders,
* then requeue. */ if ((lockres->l_blocking == DLM_LOCK_EX)
&& (lockres->l_ex_holders || lockres->l_ro_holders)) {
mlog(ML_BASTS, "lockres %s, ReQ: EX/PR Holders %u,%u\n",
lockres->l_name, lockres->l_ex_holders,
lockres->l_ro_holders); goto leave_requeue;
}
/* If it's a PR we're blocking, then only
* requeue if we've got any EX holders */ if (lockres->l_blocking == DLM_LOCK_PR &&
lockres->l_ex_holders) {
mlog(ML_BASTS, "lockres %s, ReQ: EX Holders %u\n",
lockres->l_name, lockres->l_ex_holders); goto leave_requeue;
}
/* If we get here, then we know that there are no more *incompatibleholders(andanyoneaskingforanincompatible
* lock is blocked). We can now downconvert the lock */ if (!lockres->l_ops->downconvert_worker) goto downconvert;
/* Some lockres types want to do a bit of work before *downconvertingalock.Allowthathere.Theworkerfunction *maysleep,sowesaveoffacopyofwhatwe'reblockingas
* it may change while we're not holding the spin lock. */
blocking = lockres->l_blocking;
level = lockres->l_level;
spin_unlock_irqrestore(&lockres->l_lock, flags);
if (filemap_fdatawrite(mapping)) {
mlog(ML_ERROR, "Could not sync inode %llu for downconvert!",
(unsignedlonglong)OCFS2_I(inode)->ip_blkno);
}
sync_mapping_buffers(mapping); if (blocking == DLM_LOCK_EX) {
truncate_inode_pages(mapping, 0);
} else { /* We only need to wait on the I/O if we're not also *truncatingpagesbecausetruncate_inode_pageswaits *forusabove.Wedon'ttruncatepagesifwe're *blockinganything<EXMODEbecausewewanttokeep
* them around in that case. */
filemap_fdatawait(mapping);
}
out_forget:
forget_all_cached_acls(inode);
out: return UNBLOCK_CONTINUE;
}
staticint ocfs2_ci_checkpointed(struct ocfs2_caching_info *ci, struct ocfs2_lock_res *lockres, int new_level)
{ int checkpointed = ocfs2_ci_fully_checkpointed(ci);
/* Lock quota info, this function expects at least shared lock on the quota file
* so that we can safely refresh quota info from disk. */ int ocfs2_qinfo_lock(struct ocfs2_mem_dqinfo *oinfo, int ex)
{ struct ocfs2_lock_res *lockres = &oinfo->dqi_gqlock; struct ocfs2_super *osb = OCFS2_SB(oinfo->dqi_gi.dqi_sb); int level = ex ? DLM_LOCK_EX : DLM_LOCK_PR; int status = 0;
/* On RO devices, locking really isn't needed... */ if (ocfs2_is_hard_readonly(osb)) { if (ex)
status = -EROFS; goto bail;
} if (ocfs2_mount_local(osb)) goto bail;
status = ocfs2_cluster_lock(osb, lockres, level, 0, 0); if (status < 0) {
mlog_errno(status); goto bail;
} if (!ocfs2_should_refresh_lock_res(lockres)) goto bail; /* OK, we have the lock but we need to refresh the quota info */
status = ocfs2_refresh_qinfo(oinfo); if (status)
ocfs2_qinfo_unlock(oinfo, ex);
ocfs2_complete_lock_res_refresh(lockres, status);
bail: return status;
}
int ocfs2_refcount_lock(struct ocfs2_refcount_tree *ref_tree, int ex)
{ int status; int level = ex ? DLM_LOCK_EX : DLM_LOCK_PR; struct ocfs2_lock_res *lockres = &ref_tree->rf_lockres; struct ocfs2_super *osb = lockres->l_priv;
if (ocfs2_is_hard_readonly(osb)) return -EROFS;
if (ocfs2_mount_local(osb)) return0;
status = ocfs2_cluster_lock(osb, lockres, level, 0, 0); if (status < 0)
mlog_errno(status);
return status;
}
void ocfs2_refcount_unlock(struct ocfs2_refcount_tree *ref_tree, int ex)
{ int level = ex ? DLM_LOCK_EX : DLM_LOCK_PR; struct ocfs2_lock_res *lockres = &ref_tree->rf_lockres; struct ocfs2_super *osb = lockres->l_priv;
if (!ocfs2_mount_local(osb))
ocfs2_cluster_unlock(osb, lockres, level);
}
/* Detect whether a lock has been marked as going away while *thedownconvertthreadwasprocessingotherthings.Alockcan *stillbemarkedwithOCFS2_LOCK_FREEINGafterthischeck, *butshortcircuitingherewillstillsaveussome
* performance. */
spin_lock_irqsave(&lockres->l_lock, flags); if (lockres->l_flags & OCFS2_LOCK_FREEING) goto unqueue;
spin_unlock_irqrestore(&lockres->l_lock, flags);
status = ocfs2_unblock_lock(osb, lockres, &ctl); if (status < 0)
mlog_errno(status);
if (lockres->l_flags & OCFS2_LOCK_FREEING) { /* Do not schedule a lock for downconvert when it's on *thewaytodestruction-anynodeswantingaccess
* to the resource will get it soon. */
mlog(ML_BASTS, "lockres %s won't be scheduled: flags 0x%lx\n",
lockres->l_name, lockres->l_flags); return;
}
spin_lock_irqsave(&osb->dc_task_lock, flags); /* grab this early so we know to try again if a state change and
* wake happens part-way through our work */
osb->dc_work_sequence = osb->dc_wake_sequence;
/* only quit once we've been asked to stop and there is no more
* work available */ while (!(kthread_should_stop() &&
ocfs2_downconvert_thread_lists_empty(osb))) {
spin_lock_irqsave(&osb->dc_task_lock, flags); /* make sure the voting thread gets a swipe at whatever changes
* the caller may have made to the voting state */
osb->dc_wake_sequence++;
spin_unlock_irqrestore(&osb->dc_task_lock, flags);
wake_up(&osb->dc_event);
}
Messung V0.5 in Prozent
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.0.213Bemerkung:
(vorverarbeitet am 2026-09-29)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.