/** If there are buf_pool.curr_size() per the number below pending reads, then read-aheadisnotdone:thisistopreventfloodingthebufferpoolwith
i/o-fixed buffer blocks */ #define BUF_READ_AHEAD_PEND_LIMIT 2
/** Initialize a page for read to the buffer buf_pool. If the page is (1)alreadyinbuf_pool,or (2)ifthetablespacehasbeenorisbeingdeleted, thenthisfunctiondoesnothing. Setstheio_fixflagtoBUF_IO_READandsetsanon-recursiveexclusivelock onthebufferframe.Theio-handlermusttakecarethattheflagiscleared andthelockreleasedlater. @parampage_idpageidentifier @paramzip_sizeROW_FORMAT=COMPRESSEDpagesize,or0, bitwise-ORedwith1inrecovery @paramchainbuf_pool.page_hashcellforpage_id @paramblockpreallocatedbufferblock(settonullptrifconsumed) @returnpointertotheblock @retvalnullptrincaseofanerror
@retval pointer to block | 1 if the page already exists in buf_pool */ static buf_page_t *buf_page_init_for_read(const page_id_t page_id,
ulint zip_size,
buf_pool_t::hash_chain &chain,
buf_block_t *&block) noexcept
{
buf_page_t *bpage= !zip_size || (zip_size & 1) ? &block->page : nullptr;
constexpr uint32_t READ_BUF_FIX{buf_page_t::READ_FIX + 1};
page_hash_latch &hash_lock= buf_pool.page_hash.lock_get(chain);
hash_lock.lock();
buf_page_t *hash_page= buf_pool.page_hash.get(page_id, chain); if (hash_page)
{
page_exists: /* The page is already in the buffer pool. */
ut_d(const uint32_t state=) hash_page->fix();
ut_ad(state >= buf_page_t::FREED);
hash_lock.unlock(); returnreinterpret_cast<buf_page_t*>(uintptr_t(hash_page) | 1);
}
if (UNIV_UNLIKELY(mysql_mutex_trylock(&buf_pool.mutex)))
{
hash_lock.unlock();
mysql_mutex_lock(&buf_pool.mutex);
hash_lock.lock();
hash_page= buf_pool.page_hash.get(page_id, chain); if (hash_page)
{
mysql_mutex_unlock(&buf_pool.mutex); goto page_exists;
}
}
zip_size&= ~1;
if (UNIV_LIKELY(bpage != nullptr))
{
block= nullptr; reinterpret_cast<buf_block_t*>(bpage)->
initialise(page_id, zip_size & ~1, READ_BUF_FIX); /* x_unlock() will be invoked
in buf_page_t::read_complete() by the io-handler thread. */
bpage->lock.x_lock(true); /* Insert into the hash table of file pages */
buf_pool.page_hash.append(chain, bpage);
hash_lock.unlock();
/* The block must be put to the LRU list, to the old blocks */
buf_LRU_add_block(bpage, true/* to old blocks */);
if (UNIV_UNLIKELY(zip_size))
{ /* buf_pool.mutex may be released and reacquired by buf_buddy_alloc().Wemustdeferthisoperationuntilafterthe blockdescriptorhasbeenaddedtobuf_pool.LRUand
buf_pool.page_hash. */
bpage->zip.data= static_cast<page_zip_t*>(buf_buddy_alloc(zip_size));
/* To maintain the invariant block->in_unzip_LRU_list==block->page.belongs_to_unzip_LRU() wehavetoaddthisblocktounzip_LRU
after block->page.zip.data is set. */
ut_ad(bpage->belongs_to_unzip_LRU());
buf_unzip_LRU_add_block(reinterpret_cast<buf_block_t*>(bpage), TRUE);
}
} else
{
hash_lock.unlock(); /* The compressed page must be allocated before the controlblock(bpage),inordertoavoidthe invocationofbuf_buddy_relocate_block()on
uninitialized data. */ bool lru= false; void *data= buf_buddy_alloc(zip_size, &lru);
/* If buf_buddy_alloc() allocated storage from the LRU list, itreleasedandreacquiredbuf_pool.mutex.Thus,wemust
check the page_hash again, as it may have been modified. */ if (UNIV_UNLIKELY(lru))
{
hash_page= buf_pool.page_hash.get(page_id, chain); if (UNIV_LIKELY_NULL(hash_page))
{ /* The block was added by some other thread. */
ut_d(const uint32_t state=) hash_page->fix();
ut_ad(state >= buf_page_t::FREED);
buf_buddy_free(data, zip_size);
mysql_mutex_unlock(&buf_pool.mutex); returnreinterpret_cast<buf_page_t*>(uintptr_t(hash_page) | 1);
}
}
/* Because bpage is a compressed-only block descriptor, it cannot be passedtobuf_pool.page_guess(),andthereforethereisnoriskofa afalsematch.Therefore,wecansafelyinitializebpagebefore
acquiring hash_lock. */
bpage->lock.init();
bpage->init(READ_BUF_FIX, page_id);
bpage->lock.x_lock(true);
/* The block must be put to the LRU list, to the old blocks.
The zip size is already set into the page zip */
buf_LRU_add_block(bpage, true/* to old blocks */);
}
/** Low-level function which reads a page asynchronously from a file to the bufferbuf_poolifitisnotalreadythere,inwhichcasedoesnothing. Setstheio_fixflagandsetsanexclusivelockonthebufferframe.The flagisclearedandthex-lockreleasedbyani/o-handlerthread.
@param[in]page_idpageid @param[in]zip_size0orROW_FORMAT=COMPRESSEDpagesize bitwise-ORedwith1toallocateanuncompressedframe @param[out]errnullptrforasynchronous;errorcodeforsynchronous: DB_SUCCESSifthepagewassuccessfullyread, DB_SUCCESS_LOCKED_RECiftheexistsinthepool, DB_PAGE_CORRUPTEDonpagechecksummismatch, DB_DECRYPTION_FAILEDifpagepostencryptionchecksum matchesbutafterdecryptionnormalpagechecksum doesnotmatch @param[in,out]chainbuf_pool.page_hashcellforpage_id @param[in,out]spacetablespace @param[in,out]blockpreallocatedbufferblock @param[in]thdcurrent_thdifsync @returnbuffer-fixedblock(*errmaybesettoDB_SUCCESS_LOCKED_REC) @retval-1iferr==nullptrandanasynchronousreadwassubmitted @retval-2iferr==nullptrandthepageexistsinthebufferpool
@retval nullptr if the page was not successfully read (*err will be set) */ static
buf_page_t*
buf_read_page_low( const page_id_t page_id,
ulint zip_size,
dberr_t* err,
buf_pool_t::hash_chain& chain,
fil_space_t* space,
buf_block_t*& block,
THD* thd = nullptr) noexcept
{ if (buf_dblwr.is_inside(page_id))
{
fail:
space->release(); if (err)
*err= DB_PAGE_CORRUPTED; return nullptr;
}
/** Free a buffer block if needed.
@param block block to be freed */ staticvoid buf_read_release(buf_block_t *block) noexcept
{ if (block)
{
mysql_mutex_lock(&buf_pool.mutex);
buf_LRU_block_free_non_file_page(block);
mysql_mutex_unlock(&buf_pool.mutex);
}
}
ATTRIBUTE_NOINLINE /** Free a buffer block if needed, and update the read-ahead count. @paramblockblocktobefreed @paramcountnumberofblocksthatwerereadahead
@return count*/ static size_t buf_read_release_count(buf_block_t *block, size_t count) noexcept
{ if (block || count)
{
mysql_mutex_lock(&buf_pool.mutex); if (block)
buf_LRU_block_free_non_file_page(block); if (count)
{ /* Read ahead is considered one I/O operation for the purpose of
LRU policy decision. */
buf_LRU_stat_inc_io();
buf_pool.stat.n_ra_pages_read+= count;
}
mysql_mutex_unlock(&buf_pool.mutex);
}
if (count) if (THD *thd= current_thd) if (trx_t *trx= thd_to_trx(thd)) if (ha_handler_stats *stats= trx->active_handler_stats)
stats->pages_prefetched+= count; return count;
}
/** Applies a random read-ahead in buf_pool if there are at least a threshold valueofaccessedpagesfromtherandomread-aheadarea.Doesnotreadany page,noteventheoneattheposition(space,offset),iftheread-ahead mechanismisnotactivated.NOTE:thecallingthreadmayownlatcheson pages:toavoiddeadlocksthisfunctionmustbewrittensuchthatitcannot endupwaitingfortheselatches! @param[in]page_idpageidofapagewhichthecurrentthread wantstoaccess
@return number of page read requests issued */
TRANSACTIONAL_TARGET
ulint buf_read_ahead_random(const page_id_t page_id) noexcept
{ if (!srv_random_read_ahead || page_id.space() >= SRV_TMP_SPACE_ID) /* Disable the read-ahead for temporary tablespace */ return0;
if (srv_startup_is_before_trx_rollback_phase) /* No read-ahead to avoid thread deadlocks */ return0;
if (os_aio_pending_reads_approx() >
buf_pool.curr_size() / BUF_READ_AHEAD_PEND_LIMIT) return0;
fil_space_t* space= fil_space_t::get(page_id.space()); if (!space) return0;
/* Count how many blocks in the area have been recently accessed,
that is, reside near the start of the LRU list. */
for (page_id_t i= low; i < high; ++i)
{
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(i.fold());
transactional_shared_lock_guard<page_hash_latch> g
{buf_pool.page_hash.lock_get(chain)}; if (const buf_page_t *bpage= buf_pool.page_hash.get(i, chain)) if (bpage->is_accessed() && buf_page_peek_if_young(bpage) && !--count) goto read_ahead;
}
no_read_ahead:
space->release(); return0;
read_ahead: if (space->is_stopping()) goto no_read_ahead;
/* Read all the suitable blocks within the area */
buf_block_t *block= nullptr; unsigned zip_size{space->zip_size()}; if (UNIV_LIKELY(!zip_size))
{
allocate_block: if (UNIV_UNLIKELY(!(block= buf_read_acquire()))) goto no_read_ahead;
} elseif (recv_recovery_is_on())
{
zip_size|= 1; goto allocate_block;
}
/* Read all the suitable blocks within the area */ for (page_id_t i= low; i < high; ++i)
{ if (space->is_stopping()) break;
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(i.fold());
space->reacquire(); if (reinterpret_cast<buf_page_t*>(-1) ==
buf_read_page_low(i, zip_size, nullptr, chain, space, block, nullptr))
{
count++;
ut_ad(!block); if ((UNIV_LIKELY(!zip_size) || (zip_size & 1)) &&
UNIV_UNLIKELY(!(block= buf_read_acquire()))) break;
}
}
/* Our caller should already have ensured that the page does not
exist in buf_pool.page_hash. */
buf_block_t *block= nullptr; unsigned zip_size= space->zip_size();
void buf_read_page_background(const page_id_t page_id, fil_space_t *space,
trx_t *trx) noexcept
{
ut_ad(!recv_recovery_is_on());
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(page_id.fold()); if (buf_pool.page_hash_contains(page_id, chain))
skip:
space->release(); else
{
buf_block_t *b= nullptr;
ulint zip_size{space->zip_size()}; if (UNIV_LIKELY(!zip_size) && UNIV_UNLIKELY(!(b= buf_read_acquire()))) goto skip;
buf_read_page_low(page_id, zip_size, nullptr, chain, space, b, nullptr); if (b || trx)
{
mysql_mutex_lock(&buf_pool.mutex); if (b)
buf_LRU_block_free_non_file_page(b); if (UNIV_LIKELY(trx != nullptr))
{
buf_LRU_stat_inc_io();
buf_pool.stat.n_ra_pages_read++;
}
mysql_mutex_unlock(&buf_pool.mutex);
} if (!trx); elseif (ha_handler_stats *stats= trx->active_handler_stats)
stats->pages_prefetched++; /* buf_load() invokes this with trx=nullptr. In that case, we will notupdateanystatistics;thesedeliberatepagereadsarenot partofanormalworkloadandthereforeshouldnotaffectthe
unzip_LRU heuristics. */
}
}
/** Applies linear read-ahead if in the buf_pool the page is a border page of alinearread-aheadareaandallthepagesintheareahavebeenaccessed. Doesnotreadanypageiftheread-aheadmechanismisnotactivated.Note thatthealgorithmlooksatthe'natural'adjacentsuccessorand predecessorofthepage,whichontheleaflevelofaB-treearethenext andpreviouspageinthechainofleaves.Toknowthese,thepagespecified in(space,offset)mustalreadybepresentinthebuf_pool.Thus,the naturalwaytousethisfunctionistocallitwhenapageinthebuf_pool isaccessedthefirsttime,callingthisfunctionjustafterithasbeen bufferfixed. NOTE1:asthisfunctionlooksatthenaturalpredecessorandsuccessor fieldsonthepage,whathappens,ifthesearenotinitializedtoany sensiblevalue?Noproblem,beforeapplyingread-aheadwecheckthatthe areatoreadiswithinthespanofthespace,ifnot,read-aheadisnot applied.Anuninitializedvaluemayresultinauselessreadoperation,but onlyveryimprobably. NOTE2:thecallingthreadmayownlatchesonpages:toavoiddeadlocksthis functionmustbewrittensuchthatitcannotendupwaitingforthese latches! @param[in]page_idpageid;seeNOTE3above
@return number of page read requests issued */
ulint buf_read_ahead_linear(const page_id_t page_id) noexcept
{ /* check if readahead is disabled.
Disable the read ahead logic for temporary tablespace */ if (!srv_read_ahead_threshold || page_id.space() >= SRV_TMP_SPACE_ID) return0;
if (srv_startup_is_before_trx_rollback_phase) /* No read-ahead to avoid thread deadlocks */ return0;
if (os_aio_pending_reads_approx() >
buf_pool.curr_size() / BUF_READ_AHEAD_PEND_LIMIT) return0;
/* We will check that almost all pages in the area have been accessed
in the desired order. */ constbool descending= page_id != low;
if (!descending && page_id != high_1) /* This is not a border page of the area */ return0;
fil_space_t *space= fil_space_t::get(page_id.space()); if (!space) return0;
if (high_1.page_no() > space->last_page_number())
{ /* The area is not whole. */
fail:
space->release(); return0;
}
if (trx_sys_hdr_page(page_id)) /* If it is an ibuf bitmap page or trx sys hdr, we do no
read-ahead, as that could break the ibuf page access order */ goto fail;
/* How many out of order accessed pages can we ignore
when working out the access pattern for linear readahead */
ulint count= std::min<ulint>(buf_pool_t::READ_AHEAD_PAGES -
srv_read_ahead_threshold,
uint32_t{buf_pool.read_ahead_area});
page_id_t new_low= low, new_high_1= high_1; unsigned prev_accessed= 0; for (page_id_t i= low; i <= high_1; ++i)
{
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(i.fold());
page_hash_latch &hash_lock= buf_pool.page_hash.lock_get(chain); /* It does not make sense to use transactional_lock_guard here, becausewewouldhavemanycomplexconditionsinsidethememory
transaction. */
hash_lock.lock_shared();
const buf_page_t* bpage= buf_pool.page_hash.get(i, chain); if (!bpage)
{
hash_lock.unlock_shared(); if (i == page_id) goto fail;
failed: if (--count) continue; goto fail;
} constunsigned accessed= bpage->is_accessed(); if (i == page_id)
{ /* Read the natural predecessor and successor page addresses from thepage;NOTEthatbecausethecallingthreadmayhaveanx-latch onthepage,wedonotacquireans-latchonthepage,thisisto preventdeadlocks.Thehash_lockisonlyprotectingthe
buf_pool.page_hash for page i, not the bpage contents itself. */ const byte *f= bpage->frame ? bpage->frame : bpage->zip.data;
uint32_t prev= mach_read_from_4(my_assume_aligned<4>(f + FIL_PAGE_PREV));
uint32_t next= mach_read_from_4(my_assume_aligned<4>(f + FIL_PAGE_NEXT));
hash_lock.unlock_shared(); /* The underlying file page of this buffer pool page could actually bemarkedasfreed,orareadofthepageintothebufferpoolmight beinprogress.Wemayreaduninitializeddatahere.
Suppress warnings of comparing uninitialized values. */
MEM_MAKE_DEFINED(&prev, sizeof prev);
MEM_MAKE_DEFINED(&next, sizeof next); if (prev == FIL_NULL || next == FIL_NULL) goto fail;
page_id_t id= page_id; if (descending)
{ if (id == high_1)
++id; elseif (next - 1 != page_id.page_no()) goto fail; else
id.set_page_no(prev);
} else
{ if (prev + 1 != page_id.page_no()) goto fail;
id.set_page_no(next);
}
if (id != new_low && id != new_high_1) /* This is not a border page of the area: return */ goto fail; if (new_high_1.page_no() > space->last_page_number()) /* The area is not whole */ goto fail;
} else
hash_lock.unlock_shared();
if (!accessed) goto failed; /* Note that buf_page_t::is_accessed() returns the time of the firstaccess.Ifsomeblocksoftheextentexistedinthebuffer poolatthetimeofalinearaccesspattern,thefirstaccess timesmaybenonmonotonic,eventhoughthelatestaccesstimes werelinear.Thethreshold(srv_read_ahead_factor)shouldhelpa
little against this. */ bool fail= prev_accessed &&
(descending ? prev_accessed > accessed : prev_accessed < accessed);
prev_accessed= accessed; if (fail) goto failed;
}
/* If we got this far, read-ahead can be sensible: do it */
buf_block_t *block= nullptr; unsigned zip_size{space->zip_size()}; if (UNIV_LIKELY(!zip_size))
{
allocate_block: if (UNIV_UNLIKELY(!(block= buf_read_acquire()))) goto fail;
} elseif (recv_recovery_is_on())
{
zip_size|= 1; goto allocate_block;
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.