#ifndef UNIV_INNOCHECKSUM /** Compute the number of page frames needed for buf_block_t, perinnodb_buffer_pool_extent_size. @parampsinnodb_page_size
@return number of buf_block_t frames per extent */ static constexpr uint8_t first_page(size_t ps)
{ return uint8_t(innodb_buffer_pool_extent_size / ps -
innodb_buffer_pool_extent_size / (ps + sizeof(buf_block_t)));
}
/** Compute the number of bytes needed for buf_block_t, perinnodb_buffer_pool_extent_size. @parampsinnodb_page_size
@return number of buf_block_t frames per extent */ static constexpr size_t first_frame(size_t ps)
{ return first_page(ps) * ps;
}
/** Compute the number of pages per innodb_buffer_pool_extent_size. @parampsinnodb_page_size
@return number of buf_block_t frames per extent */ static constexpr uint16_t pages(size_t ps)
{ return uint16_t(innodb_buffer_pool_extent_size / ps - first_page(ps));
}
/** The byte offset of the first page frame in a buffer pool extent
of innodb_buffer_pool_extent_size bytes */ static constexpr size_t first_frame_in_extent[]=
{
first_frame(4096), first_frame(8192), first_frame(16384),
first_frame(32768), first_frame(65536)
};
/** The position offset of the first page frame in a buffer pool extent
of innodb_buffer_pool_extent_size bytes */ static constexpr uint8_t first_page_in_extent[]=
{
first_page(4096), first_page(8192), first_page(16384),
first_page(32768), first_page(65536)
};
/** Number of pages per buffer pool extent
of innodb_buffer_pool_extent_size bytes */ static constexpr size_t pages_in_extent[]=
{
pages(4096), pages(8192), pages(16384), pages(32768), pages(65536)
};
void buf_inc_get() noexcept
{ if (THD *thd= current_thd) if (trx_t *trx= thd_to_trx(thd))
buf_inc_get(trx);
}
# ifdef SUX_LOCK_GENERIC void page_hash_latch::read_lock_wait() noexcept
{ /* First, try busy spinning for a while. */ for (auto spin= srv_n_spin_wait_rounds; spin--; )
{
LF_BACKOFF(); if (read_trylock()) return;
} /* Fall back to yielding to other threads. */ do
std::this_thread::yield(); while (!read_trylock());
}
/* First, try busy spinning for a while. */ for (auto spin= srv_n_spin_wait_rounds; spin--; )
{ if (write_lock_poll()) return;
LF_BACKOFF();
}
/* Fall back to yielding to other threads. */ do
std::this_thread::yield(); while (!write_lock_poll());
} # endif
/** Number of attempts made to read in a page in the buffer pool */
constexpr ulint BUF_PAGE_READ_MAX_RETRIES= 100; /** The maximum portion of the buffer pool that can be used for the
read-ahead buffer. (Divide buf_pool size by this amount) */
constexpr uint32_t BUF_READ_AHEAD_PORTION= 32;
/** A 64KiB buffer of NUL bytes, for use in assertions and checks, anddummydefaultvaluesofinstantlydroppedcolumns. Initially,BLOBfieldreferencesaresettoNULbytes,in
dtuple_convert_big_rec(). */ const byte *field_ref_zero;
/** The InnoDB buffer pool */
buf_pool_t buf_pool;
#ifdef UNIV_DEBUG /** This is used to insert validation operations in execution
in the debug version */ static Atomic_counter<size_t> buf_dbg_counter; #endif/* UNIV_DEBUG */
/** Macro to determine whether the read of write counter is used depending
on the io_type */ #define MONITOR_RW_COUNTER(read, counter) \
(read ? (counter##_READ) : (counter##_WRITTEN))
/** Decrypt a page for temporary tablespace. @param[in,out]tmp_frameTemporarybuffer @param[in]src_framePagetodecrypt
@return true if temporary tablespace decrypted, false if not */ staticbool buf_tmp_page_decrypt(byte* tmp_frame, byte* src_frame)
{ if (buf_is_zeroes(span<const byte>(src_frame, srv_page_size))) { returntrue;
}
/* read space & lsn */
uint header_len = FIL_PAGE_FILE_FLUSH_LSN_OR_KEY_VERSION;
/* Copy FIL page header, it is not encrypted */
memcpy(tmp_frame, src_frame, header_len);
if (id.page_no() == 0) { /* File header pages are not encrypted/compressed */ return (true);
}
buf_tmp_buffer_t* slot;
if (id.space() == SRV_TMP_SPACE_ID
&& innodb_encrypt_temporary_tables) {
slot = buf_pool.io_buf_reserve(false);
slot->allocate(); bool ok = buf_tmp_page_decrypt(slot->crypt_buf, dst_frame);
slot->release(); return ok;
}
/* Page is encrypted if encryption information is found from tablespaceandpagecontainsusedkey_version.Thisistrue
also for pages first compressed and then encrypted. */
if (checksum_field1 != checksum_field2) { returnfalse;
}
return checksum_field1 == crc32;
}
#ifndef UNIV_INNOCHECKSUM /** Check whether a page is newer than the durable LSN. @paramcheck_lsnwhethertochecktheLSN @paramread_bufpageframe
@return whether the FIL_PAGE_LSN is invalid (ahead of the durable LSN) */ staticbool buf_page_check_lsn(bool check_lsn, const byte *read_buf) noexcept
{ if (!check_lsn) returnfalse; /* A page may not be read before it is written, and it may not be writtenbeforethecorrespondingloghasbeendurablywritten.
Hence, we refer to the current durable LSN here */
lsn_t current_lsn= log_sys.get_flushed_lsn(std::memory_order_relaxed); if (UNIV_UNLIKELY(current_lsn == log_sys.FIRST_LSN) &&
srv_force_recovery == SRV_FORCE_NO_LOG_REDO) returnfalse; const lsn_t page_lsn= mach_read_from_8(read_buf + FIL_PAGE_LSN);
if (UNIV_LIKELY(current_lsn >= page_lsn)) returnfalse;
sql_print_error("InnoDB: Page " "[page id: space=" UINT32PF ", page number=" UINT32PF "]" " log sequence number " LSN_PF " is in the future! Current system log sequence number "
LSN_PF ".",
space_id, page_no, page_lsn, current_lsn);
if (srv_force_recovery) returnfalse;
sql_print_error("InnoDB: Your database may be corrupt or" " you may have copied the InnoDB" " tablespaces but not the log. %s",
FORCE_RECOVERY_MSG);
returntrue;
} #endif
/** Check if a buffer is all zeroes. @param[in]bufdatatocheck
@return whether the buffer is all zeroes */ bool buf_is_zeroes(span<const byte> buf) noexcept
{
ut_ad(buf.size() <= UNIV_PAGE_SIZE_MAX); return memcmp(buf.data(), field_ref_zero, buf.size()) == 0;
}
/** Check if a page is corrupt. @paramcheck_lsnwhetherFIL_PAGE_LSNshouldbechecked @paramread_bufdatabasepage @paramfsp_flagscontentsofFIL_SPACE_FLAGS
@return whether the page is corrupted */
buf_page_is_corrupted_reason
buf_page_is_corrupted(bool check_lsn, const byte *read_buf, uint32_t fsp_flags)
noexcept
{ if (fil_space_t::full_crc32(fsp_flags)) { bool compressed = false, corrupted = false; const uint size = buf_page_full_crc32_size(
read_buf, &compressed, &corrupted); if (corrupted) { return CORRUPTED_OTHER;
} const byte* end = read_buf + (size - FIL_PAGE_FCRC32_CHECKSUM);
uint crc32 = mach_read_from_4(end);
if (!zip_size
&& memcmp_aligned<4>(read_buf + FIL_PAGE_LSN + 4,
read_buf + srv_page_size
- FIL_PAGE_END_LSN_OLD_CHKSUM + 4, 4)) { /* Stored log sequence numbers at the start and the end
of page do not match */
return CORRUPTED_OTHER;
}
/* Check whether the checksum fields have correct values */
if (zip_size) { if (!page_zip_verify_checksum(read_buf, zip_size)) { return CORRUPTED_OTHER;
} goto check_lsn;
}
/* A page filled with NUL bytes is considered not corrupted. BeforeMariaDBServer10.1.25(MDEV-12113)or10.2.2(orMySQL5.7), theFIL_PAGE_FILE_FLUSH_LSNfieldmayhavebeenwrittennonzero forthefirstpageofeachfileofthesystemtablespace. Wewanttoignoreitforthesystemtablespace,butbecause wedonotknowtheexpectedtablespacehere,weignorethe fieldforalldatafiles,exceptfor
innodb_checksum_algorithm=full_crc32 which we handled above. */ if (!checksum_field1 && !checksum_field2) { /* Checksum fields can have valid value as zero. Ifthepageisnotemptythendothechecksum
calculation for the page. */ bool all_zeroes = true; for (size_t i = 0; i < srv_page_size; i++) { #ifndef UNIV_INNOCHECKSUM if (i == FIL_PAGE_FILE_FLUSH_LSN_OR_KEY_VERSION) {
i += 8;
} #endif if (read_buf[i]) {
all_zeroes = false; break;
}
}
if (all_zeroes) { return NOT_CORRUPTED;
}
}
#ifndef UNIV_INNOCHECKSUM switch (srv_checksum_algorithm) { case SRV_CHECKSUM_ALGORITHM_STRICT_FULL_CRC32: case SRV_CHECKSUM_ALGORITHM_STRICT_CRC32: #endif/* !UNIV_INNOCHECKSUM */ if (!buf_page_is_checksum_valid_crc32(read_buf,
checksum_field1,
checksum_field2)) { return CORRUPTED_OTHER;
} goto check_lsn; #ifndef UNIV_INNOCHECKSUM default: if (checksum_field1 == BUF_NO_CHECKSUM_MAGIC
&& checksum_field2 == BUF_NO_CHECKSUM_MAGIC) { goto check_lsn;
}
for (auto trig= std::begin(m_triggers); trig!= std::end(m_triggers); ++trig)
{ if ((m_fds[m_num_fds].fd=
open(memcgroup.c_str(), O_RDWR | O_NONBLOCK | O_CLOEXEC)) < 0)
{ /* User can't do anything about it, no point giving warning */
shutdown(); returnfalse;
}
my_register_filename(m_fds[m_num_fds].fd, memcgroup.c_str(), FILE_BY_OPEN, 0, MYF(0));
ssize_t slen= strlen(*trig); if (write(m_fds[m_num_fds].fd, *trig, slen) < slen)
{ /* we may fail this one, but continue to the next */
my_close(m_fds[m_num_fds].fd, MYF(MY_WME)); continue;
}
m_fds[m_num_fds].events= POLLPRI;
m_num_fds++;
} if (m_num_fds < 1) returnfalse;
if ((m_event_fd= eventfd(0, EFD_CLOEXEC|EFD_NONBLOCK)) == -1)
{ /* User can't do anything about it, no point giving warning */
shutdown(); returnfalse;
}
my_register_filename(m_event_fd, "mem_pressure_eventfd", FILE_BY_DUP, 0, MYF(0));
m_fds[m_num_fds].fd= m_event_fd;
m_fds[m_num_fds].events= POLLIN;
m_num_fds++;
m_thd= std::thread(pressure_routine, this);
sql_print_information("InnoDB: Initialized memory pressure event listener"); returntrue;
}
void shutdown()
{ /* m_event_fd is in this list */ while (m_num_fds)
{
m_num_fds--;
my_close(m_fds[m_num_fds].fd, MYF(MY_WME));
m_fds[m_num_fds].fd= -1;
}
m_event_fd= -1;
}
ulonglong last= microsecond_interval_timer() - max_interval_us; while (!m->m_abort)
{ if (poll(&m->m_fds[0], m->m_num_fds, -1) < 0)
{ if (errno == EINTR) continue; else break;
} if (m->m_abort) break;
for (pollfd &p : st_::span<pollfd>(m->m_fds, m->m_num_fds))
{ if (p.revents & POLLPRI)
{
ulonglong now= microsecond_interval_timer(); if ((now - last) > max_interval_us)
{
last= now;
buf_pool.garbage_collect();
}
}
#ifdef UNIV_DEBUG if (p.revents & POLLIN)
{
uint64_t u; /* we haven't aborted, so this must be a debug trigger */ if (read(p.fd, &u, sizeof(u)) >=0)
buf_pool.garbage_collect();
} #endif
}
}
m->shutdown();
@return number of errors found in madvise() calls */
MY_ATTRIBUTE((used)) int buf_pool_t::madvise_do_dump() noexcept
{ int ret= 0;
/* mirrors allocation in log_t::create() */ if (log_sys.buf) {
ret += madvise(log_sys.buf, log_sys.buf_size, MADV_DODUMP);
ret += madvise(log_sys.flush_buf, log_sys.buf_size,
MADV_DODUMP);
}
if (UNIV_LIKELY(!n_blocks_to_withdraw) || !withdraw(*b))
{ /* No adaptive hash index entries may point to a free block. */
assert_block_ahi_empty(reinterpret_cast<buf_block_t*>(b));
b->set_state(buf_page_t::MEMORY);
b->set_os_used(); returnreinterpret_cast<buf_block_t*>(b);
}
}
return nullptr;
}
/** Create the hash table.
@param n the lower bound of n_cells */ void buf_pool_t::page_hash_table::create(ulint n) noexcept
{
n_cells= ut_find_prime(n); const size_t size= MY_ALIGN(pad(n_cells) * sizeof *array,
CPU_LEVEL1_DCACHE_LINESIZE); void *v= aligned_malloc(size, CPU_LEVEL1_DCACHE_LINESIZE);
memset_aligned<CPU_LEVEL1_DCACHE_LINESIZE>(v, 0, size);
array= static_cast<hash_chain*>(v);
}
/* Let us aim for 128 GiB (a quarter of the 512 GiB), or the
initial innodb_buffer_pool_size, whichever is greater. */
size_in_bytes_max= std::max(size_t(1ULL << 37), size_in_bytes_requested); goto init;
} #endif else goto oom;
}
/** Relocate a ROW_FORMAT=COMPRESSED block in the LRU list and buf_pool.page_hash. Thecallermustrelocatebpage->list. @parambpageROW_FORMAT=COMPRESSEDonlyblock
@param dpage destination control block */ staticvoid buf_relocate(buf_page_t *bpage, buf_page_t *dpage) noexcept
{ const page_id_t id{bpage->id()};
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(id.fold());
ut_ad(!bpage->frame);
mysql_mutex_assert_owner(&buf_pool.mutex);
ut_ad(mach_read_from_4(bpage->zip.data + FIL_PAGE_OFFSET) == id.page_no());
ut_ad(buf_pool.page_hash.lock_get(chain).is_write_locked());
ut_ad(bpage == buf_pool.page_hash.get(id, chain));
ut_d(constauto state= bpage->state());
ut_ad(state >= buf_page_t::FREED);
ut_ad(state <= buf_page_t::READ_FIX);
ut_ad(bpage->lock.is_write_locked()); constauto frame= dpage->frame;
ut_ad(frame == reinterpret_cast<buf_block_t*>(dpage)->frame_address());
dpage->lock.free(); new (dpage) buf_page_t(*bpage);
dpage->frame= frame;
/* Important that we adjust the hazard pointer before
removing bpage from LRU list. */ if (buf_page_t *b= buf_pool.LRU_remove(bpage))
UT_LIST_INSERT_AFTER(buf_pool.LRU, b, dpage); else
UT_LIST_ADD_FIRST(buf_pool.LRU, dpage);
if (UNIV_UNLIKELY(buf_pool.LRU_old == bpage))
{
buf_pool.LRU_old= dpage; #ifdef UNIV_LRU_DEBUG /* buf_pool.LRU_old must be the first item in the LRU list
whose "old" flag is set. */
ut_a(buf_pool.LRU_old->old);
ut_a(!UT_LIST_GET_PREV(LRU, buf_pool.LRU_old) ||
!UT_LIST_GET_PREV(LRU, buf_pool.LRU_old)->old);
ut_a(!UT_LIST_GET_NEXT(LRU, buf_pool.LRU_old) ||
UT_LIST_GET_NEXT(LRU, buf_pool.LRU_old)->old);
} else
{ /* Check that the "old" flag is consistent in
the block and its neighbours. */
dpage->set_old(dpage->is_old()); #endif/* UNIV_LRU_DEBUG */
}
/** Mark the page status as FREED for the given tablespace and page number. @param[in,out]spacetablespace @param[in]pagepagenumber
@param[in,out] mtr mini-transaction */
TRANSACTIONAL_TARGET void buf_page_free(fil_space_t *space, uint32_t page, mtr_t *mtr)
{
ut_ad(mtr);
ut_ad(mtr->is_active());
if (srv_immediate_scrub_data_uncompressed #ifdefined HAVE_FALLOC_PUNCH_HOLE_AND_KEEP_SIZE || defined _WIN32
|| space->is_compressed() #endif
)
mtr->add_freed_offset(space, page);
buf_inc_get(); const page_id_t page_id(space->id, page);
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(page_id.fold());
uint32_t fix;
buf_block_t *block;
{
transactional_shared_lock_guard<page_hash_latch> g
{buf_pool.page_hash.lock_get(chain)};
block= reinterpret_cast<buf_block_t*>
(buf_pool.page_hash.get(page_id, chain)); if (!block || !block->page.frame) /* FIXME: convert ROW_FORMAT=COMPRESSED, without buf_zip_decompress() */ return; /* To avoid a deadlock with buf_LRU_free_page() of some other page andbuf_page_write_complete()ofthispage,wemustnotwaitfora
page latch while holding a page_hash latch. */
fix= block->page.fix();
}
if (UNIV_UNLIKELY(fix < buf_page_t::UNFIXED))
{
block->page.unfix(); return;
}
constbool got_s_latch= bpage->lock.s_lock_try();
hash_lock.unlock_shared(); if (UNIV_LIKELY(got_s_latch))
{
ut_ad(!bpage->is_read_fixed()); break;
}
/* We may fail to acquire bpage->lock because a read is holding an exclusivelatchonthisblockandeitherinprogressorinvoking buf_pool_t::corrupted_evict().
switch (fil_page_get_type(frame)) { case FIL_PAGE_INDEX: case FIL_PAGE_RTREE: if (page_zip_decompress(&block->page.zip,
block->page.frame, TRUE)) {
func_exit: if (space) {
space->release();
} returntrue;
}
ib::error() << "Unable to decompress "
<< (space ? space->chain.start->name : "")
<< block->page.id(); goto err_exit; case FIL_PAGE_TYPE_ALLOCATED: case FIL_PAGE_INODE: case FIL_PAGE_IBUF_BITMAP: case FIL_PAGE_TYPE_FSP_HDR: case FIL_PAGE_TYPE_XDES: case FIL_PAGE_TYPE_ZBLOB: case FIL_PAGE_TYPE_ZBLOB2: /* Copy to uncompressed storage. */
memcpy(block->page.frame, frame, block->zip_size()); goto func_exit;
}
/* Wait for b->unfix() in any other threads. */
uint32_t state= b->state();
ut_ad(buf_page_t::buf_fix_count(state));
ut_ad(!buf_page_t::is_freed(state));
switch (state) { case buf_page_t::UNFIXED + 1: case buf_page_t::REINIT + 1: break; default:
ut_ad(state < buf_page_t::READ_FIX);
/* Ensure that another buf_page_get_low() or buf_page_t::page_fix() willwaitforblock->page.lock.x_unlock().buf_relocate()will
copy the state from b to block and replace b with block in page_hash. */
b->set_state(buf_page_t::READ_FIX);
if (UNIV_UNLIKELY(!b->frame))
{
unzip: if (b->lock.x_lock_try()); elseif (c == FIX_NOWAIT) goto would_block; else goto wait_for_unzip;
buf_block_t *block= unzip(b, chain); if (!block) goto corrupted;
b= &block->page;
b->lock.x_unlock();
}
returnreinterpret_cast<buf_block_t*>(b);
}
hash_lock.unlock_shared();
if (c == FIX_NOWAIT) returnreinterpret_cast<buf_block_t*>(-1);
buf_block_t *block= buf_read_page(id, err, chain); if (!block) return nullptr;
buf_read_ahead_random(id); if (err)
{
ut_ad(*err == DB_SUCCESS || *err == DB_SUCCESS_LOCKED_REC);
*err= DB_SUCCESS;
} if (UNIV_UNLIKELY(!block->page.frame))
{
b= &block->page; goto unzip;
} return block;
}
}
uint32_t buf_pool_t::page_guess(buf_block_t *b, page_hash_latch &latch, const page_id_t id) noexcept
{ /* On at least two Intel Xeon of different generation, it turns out
that transactional_shared_lock_guard would perform worse here. */
latch.lock_shared(); /* shrunk() made the memory inaccessible. */ if (UNIV_UNLIKELY(reinterpret_cast<char*>(b) >= memory + size_in_bytes))
{
latch.unlock_shared(); return0;
} /* This synchronizes with buf_page_t::init() */
uint32_t state{b->page.zip.fix.load(std::memory_order_acquire)}; const page_id_t block_id{b->page.id()};
if (id == block_id)
{ /* Ignore guesses that point to read-fixed blocks. We can only
avoid a race condition by looking up the block via page_hash. */ if ((state >= buf_page_t::FREED && state < buf_page_t::READ_FIX) ||
state >= buf_page_t::WRITE_FIX)
state= b->page.fix(); else
state= 0;
ut_ad(b->page.frame);
} else
state= 0;
latch.unlock_shared(); return state;
}
/** Low level function used to get access to a database page. @param[in]page_idpageid @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]rw_latchlatchmode @param[in]guessguessedblockorNULL @param[in]modeBUF_GET,BUF_GET_IF_IN_POOL, orBUF_PEEK_IF_IN_POOL @param[in]mtrmini-transaction @param[out]errDB_SUCCESSorerrorcode @returnpointertotheblock
@retval nullptr if the block is corrupted or unavailable */
buf_block_t*
buf_page_get_gen( const page_id_t page_id,
ulint zip_size,
rw_lock_type_t rw_latch,
buf_block_t* guess,
ulint mode,
mtr_t* mtr,
dberr_t* err) noexcept
{
ulint retries = 0;
/* BUF_GET_RECOVER is only used by recv_sys_t::recover(), whichmustbeinvokedduringearlyserverstartupwhencrash recoverymaybeinprogress.Theonlycasewhenitmaybe invokedoutsiderecoveryiswhendict_create()hasinitialized anewdatabaseandisinvokingdict_boot().Inthiscase,the LSNwillbesmall.Attheendofabootstrap,theshutdownLSN wouldtypicallybearound60000withthedefault innodb_undo_tablespaces=3,andlessthan110000withthemaximum
innodb_undo_tablespaces=127. */
ut_d(externbool ibuf_upgrade_was_needed;)
ut_ad(mode == BUF_GET_RECOVER
? recv_recovery_is_on() || log_get_lsn() < 120000
|| log_get_lsn() == recv_sys.lsn + SIZE_OF_FILE_CHECKPOINT
|| ibuf_upgrade_was_needed
: !recv_recovery_is_on() || recv_sys.after_apply);
ut_ad(mtr->is_active());
if (err) {
*err = DB_SUCCESS;
}
#ifdef UNIV_DEBUG switch (mode) { default:
ut_ad(mode == BUF_PEEK_IF_IN_POOL); break; case BUF_GET_POSSIBLY_FREED: case BUF_GET_IF_IN_POOL: /* The caller may pass a dummy page size,
because it does not really matter. */ break; case BUF_GET_RECOVER: case BUF_GET:
ut_ad(!mtr->is_freeing_tree());
fil_space_t* s = fil_space_get(page_id.space());
ut_ad(s);
ut_ad(s->zip_size() == zip_size);
} #endif/* UNIV_DEBUG */
/* A memory transaction would frequently be aborted here. */
hash_lock.lock_shared();
block = reinterpret_cast<buf_block_t*>(
buf_pool.page_hash.get(page_id, chain)); if (UNIV_LIKELY(block != nullptr)) {
state = block->page.fix();
hash_lock.unlock_shared(); goto got_block;
}
hash_lock.unlock_shared();
/* Page not in buf_pool: needs to be read from file */ switch (mode) { case BUF_GET_IF_IN_POOL: case BUF_PEEK_IF_IN_POOL: break; default:
block = buf_read_page(page_id, err, chain); if (!block) { break;
} elseif (err) {
*err = DB_SUCCESS;
}
got_block:
state++; if (state > buf_page_t::READ_FIX && state < buf_page_t::WRITE_FIX) { if (mode == BUF_PEEK_IF_IN_POOL) {
ignore_block:
block->unfix();
ignore_unfixed:
ut_ad(mode == BUF_GET_POSSIBLY_FREED
|| mode == BUF_PEEK_IF_IN_POOL); if (err) {
*err = DB_CORRUPTION;
} return nullptr;
}
if (UNIV_UNLIKELY(!block->page.frame)) { goto wait_for_unzip;
} /* A read-fix is released after block->page.lock inbuf_page_t::read_complete()or buf_pool_t::corrupted_evict(),or
after buf_zip_decompress() in this function. */
block->page.read_wait(trx);
state = block->page.state();
if (UNIV_UNLIKELY(state < buf_page_t::UNFIXED)) { const page_id_t id{block->page.id()};
block->page.unfix();
block->page.lock.s_unlock();
if (UNIV_UNLIKELY(id == page_id)) { /* The page read was completed, and anotherthreadmarkedthepageasfree
while we were waiting. */ goto ignore_unfixed;
}
ut_ad(id == page_id_t{~0ULL});
if (++retries < BUF_PAGE_READ_MAX_RETRIES) { goto loop;
}
if (err) {
*err = DB_PAGE_CORRUPTED;
}
return nullptr;
}
ut_ad(block->page.id() == page_id);
if (UNIV_LIKELY(state > buf_page_t::UNFIXED
&& block->page.frame)) { switch (rw_latch) { bool nowait; case RW_NO_LATCH: break; default:
nowait = block->page.lock.s_x_upgrade(); if (rw_latch == RW_SX_LATCH) {
block->page.lock.x_u_downgrade();
} else {
ut_ad(rw_latch == RW_X_LATCH);
} if (!nowait) { goto latch_waited;
} else {
ut_ad(state < buf_page_t::READ_FIX);
} /* fall through */ case RW_S_LATCH:
mtr->memo_push(block,
mtr_memo_type_t(rw_latch)); goto latched;
}
}
block->page.lock.s_unlock();
} else {
not_read_fixed:
ut_ad(state > buf_page_t::FREED);
ut_ad(state < buf_page_t::READ_FIX
|| state > buf_page_t::WRITE_FIX); if (UNIV_UNLIKELY(!block->page.frame
&& mode == BUF_PEEK_IF_IN_POOL)) { /* The BUF_PEEK_IF_IN_POOL mode is mainly used fordroppinganadaptivehashindex.There cannotbeanadaptivehashindexfora
compressed-only page. */ goto ignore_block;
}
}
if (UNIV_UNLIKELY(state < buf_page_t::UNFIXED)) { goto ignore_block;
}
ut_ad((~buf_page_t::LRU_MASK) & state);
ut_ad(state > buf_page_t::WRITE_FIX || state < buf_page_t::READ_FIX);
if (UNIV_UNLIKELY(!block->page.frame)) { if (!block->page.lock.x_lock_try()) {
wait_for_unzip: /* The page is being read or written, or
another thread is executing buf_pool.unzip() on it. */
block->page.unfix();
std::this_thread::sleep_for(
std::chrono::microseconds(100)); goto loop;
}
block = buf_pool.unzip(&block->page, chain);
if (!block) { goto ignore_unfixed;
}
block->page.lock.x_unlock();
}
#ifdef UNIV_DEBUG if (!(++buf_dbg_counter % 5771)) buf_pool.validate(); #endif/* UNIV_DEBUG */
/* The state = block->page.state() may be stale at this point, andinfact,atanypointoftimeifweconsiderits buffer-fixcomponent.Iftheblockisbeingreadintothe bufferpool,itispossiblethatbuf_page_t::read_complete() willinvokebuf_pool_t::corrupted_evict()andtherefore invalidateit(invokebuf_page_t::set_corrupt_id()andsetthe statetoFREED).Therefore,afteracquiringthepagelatchwe
must recheck the state. */
switch (rw_latch) { case RW_NO_LATCH:
mtr->memo_push(block, MTR_MEMO_BUF_FIX); return block; case RW_S_LATCH:
block->page.lock.s_lock(); break; case RW_SX_LATCH:
block->page.lock.u_lock();
ut_ad(!block->page.is_io_fixed()); break; default:
ut_ad(rw_latch == RW_X_LATCH); if (block->page.lock.x_lock_upgraded()) {
ut_ad(block->page.id() == page_id);
block->unfix(); return mtr->page_lock_upgrade(*block);
}
}
latch_waited:
mtr->memo_push(block, mtr_memo_type_t(rw_latch));
state = block->page.state();
if (UNIV_UNLIKELY(state < buf_page_t::UNFIXED)) {
mtr->release_last_page(); goto ignore_unfixed;
}
ut_ad(state > buf_page_t::FREED);
ut_ad(state < buf_page_t::READ_FIX); /* In addition to our buffer-fix, there may be another that is heldbyaconcurrentIORequest::read_complete()thathad releasedthebpage->lockinbpage->read_complete(...)butnot yetinvokedbpage->unfix().Thisshouldonlybeduetoan asynchronousread-aheadforapagethatwasactuallymarkedas
freed in the underlying data file. */
ut_ad(bpage->buf_fix_count(state) <= 2);
if (state < buf_page_t::UNFIXED)
bpage->set_reinit(buf_page_t::FREED); else
bpage->set_reinit(state & buf_page_t::LRU_MASK);
if (UNIV_LIKELY(bpage->frame != nullptr))
{
mysql_mutex_unlock(&buf_pool.mutex);
buf_block_t *block= reinterpret_cast<buf_block_t*>(bpage);
ut_ad(bpage->frame == block->frame_address());
mtr->memo_push(block, MTR_MEMO_PAGE_X_FIX); #ifdef BTR_CUR_HASH_ADAPT
drop_hash_entry= block->index; #endif
} else
{
page_hash_latch &hash_lock= buf_pool.page_hash.lock_get(chain); for (;;)
{
hash_lock.lock();
state= bpage->state(); if ((state & ~buf_page_t::LRU_MASK) == 1) break; /* Wait for a concurrent IORequest::read_complete() to invokebpage->unfix(),foranunnecessaryread-aheadof
a freed page. */
ut_ad((state & ~buf_page_t::LRU_MASK) == 2);
hash_lock.unlock();
}
/* The block must be put to the LRU list */
buf_LRU_add_block(bpage, false);
if (UNIV_UNLIKELY(zip_size))
{
bpage->zip.data= buf_buddy_alloc(zip_size);
/* To maintain the invariant block->in_unzip_LRU_list == block->page.belongs_to_unzip_LRU()wehavetoaddthis
block to unzip_LRU after block->page.zip.data is set. */
ut_ad(bpage->belongs_to_unzip_LRU());
buf_unzip_LRU_add_block(reinterpret_cast<buf_block_t*>(bpage), FALSE);
}
/* FIL_PAGE_FILE_FLUSH_LSN_OR_KEY_VERSION is only used on the followingpages: (1)ThefirstpageoftheInnoDBsystemtablespace(page0:0) (2)FIL_RTREE_SPLIT_SEQ_NUMonR-treepages
(3) key_version on encrypted pages (not page 0:0) */
/** Check if the encrypted page is corrupted for the full crc32 format. @param[in]space_idpagebelongstospaceid @param[in]dpage @param[in]is_compressedcompressedpage
@return true if page is corrupted or false if it isn't */ staticbool buf_page_full_crc32_is_corrupted(ulint space_id, const byte* d, bool is_compressed) noexcept
{ if (space_id != mach_read_from_4(d + FIL_PAGE_SPACE_ID)) returntrue;
/** Check if page is maybe compressed, encrypted or both when we encounter corruptedpage.Notethatwecan'tbe100%sureifpageiscorrupted ordecrypt/decompressjustfailed. @param[in,out]bpagepage @param[in]nodedatafile @returnwhethertheoperationsucceeded @retvalDB_SUCCESSifpagehasbeenreadandisnotcorrupted @retvalDB_PAGE_CORRUPTEDifpagebasedonchecksumcheckiscorrupted @retvalDB_CORRUPTIONifthepageLSNisinthefuture @retvalDB_DECRYPTION_FAILEDifpagepostencryptionchecksummatchesbut
after decryption normal page checksum does not match. */ static dberr_t buf_page_check_corrupt(buf_page_t *bpage, const fil_node_t &node)
{
ut_ad(node.space->referenced());
/* In buf_decrypt_after_read we have either decrypted the page if pagepostencryptionchecksummatchesandusedkey_idisfound fromtheencryptionplugin.Ifchecksumdidnotmatchpagewas notdecryptedanditcouldbeeitherencryptedandcorrupted orcorruptedorgoodpage.Ifwedecrypted,therepagecould
still be corrupted if used key does not match. */ constbool seems_encrypted = !node.space->full_crc32() && key_version
&& node.space->crypt_data
&& node.space->crypt_data->type != CRYPT_SCHEME_UNENCRYPTED;
ut_ad(!node.space->is_temporary() || node.space->full_crc32());
/* If traditional checksums match, we assume that page is
not anymore encrypted. */ if (node.space->full_crc32()
&& !buf_is_zeroes(span<const byte>(dst_frame,
node.space->physical_size()))
&& (key_version || node.space->is_compressed()
|| node.space->is_temporary())) { if (buf_page_full_crc32_is_corrupted(
bpage->id().space(), dst_frame,
node.space->is_compressed())) {
err = DB_PAGE_CORRUPTED;
}
} else { switch (buf_page_is_corrupted(true, dst_frame,
node.space->flags)) { case NOT_CORRUPTED: break; case CORRUPTED_OTHER:
err = DB_PAGE_CORRUPTED; break; case CORRUPTED_FUTURE_LSN:
err = DB_CORRUPTION; break;
}
}
if (read_id == expected_id); elseif (read_id == page_id_t(0, 0))
{ /* This is likely an uninitialized (all-zero) page. */
err= DB_FAIL; goto release_page;
} elseif (!node.space->full_crc32() &&
page_id_t(0, read_id.page_no()) == expected_id) /* FIL_PAGE_SPACE_ID was written as garbage in the system tablespace
before MySQL 4.1.1, which introduced innodb_file_per_table. */; elseif (node.space->full_crc32() &&
*reinterpret_cast<const uint32_t*>
(&read_frame[FIL_PAGE_FCRC32_KEY_VERSION]) &&
node.space->crypt_data &&
node.space->crypt_data->type != CRYPT_SCHEME_UNENCRYPTED)
{
err= DB_DECRYPTION_FAILED; goto release_page;
} else
{
sql_print_error("InnoDB: Space id and page no stored in the page," " read in from %s are " "[page id: space=" UINT32PF ", page number=" UINT32PF "], should be " "[page id: space=" UINT32PF ", page number=" UINT32PF "]",
node.name,
read_id.space(), read_id.page_no(),
expected_id.space(), expected_id.page_no());
err= DB_FAIL; goto release_page;
}
}
if (belongs_to_unzip_LRU())
{
buf_pool.n_pend_unzip++; auto ok= buf_zip_decompress(reinterpret_cast<buf_block_t*>(this), false);
buf_pool.n_pend_unzip--;
if (!ok)
{
err= DB_PAGE_CORRUPTED; goto database_corrupted_compressed;
}
}
err= buf_page_check_corrupt(this, node); if (UNIV_UNLIKELY(err != DB_SUCCESS))
{
database_corrupted: if (belongs_to_unzip_LRU())
database_corrupted_compressed:
memset_aligned<UNIV_PAGE_SIZE_MIN>(frame, 0, srv_page_size);
release_page: if (recovery && node.space->full_crc32() && node.space->crypt_data &&
recv_sys.dblwr.find_deferred_page(node, id().page_no(), const_cast<byte*>(read_frame))) /* Recovered from the doublewrite buffer */
err= DB_SUCCESS; else
{ if (recovery && recv_sys.free_corrupted_page(expected_id, node)); elseif (err == DB_FAIL) /* We already output a more specific message. */
err= DB_PAGE_CORRUPTED; else
{
sql_print_error("InnoDB: Failed to read page " UINT32PF " from file '%s': %s", expected_id.page_no(),
node.name, ut_strerr(err));
buf_page_print(read_frame, zip_size());
if (node.space->set_corrupted() &&
!is_predefined_tablespace(node.space->id))
sql_print_information("InnoDB: You can use CHECK TABLE to scan" " your table for corruption. %s",
FORCE_RECOVERY_MSG);
}
if (UNIV_UNLIKELY(MONITOR_IS_ON(MONITOR_MODULE_BUF_PAGE)))
buf_page_monitor(*this, true);
DBUG_PRINT("ib_buf", ("read page %u:%u", id().space(), id().page_no()));
lock.x_unlock(true);
return DB_SUCCESS;
}
#ifdef BTR_CUR_HASH_ADAPT /** Clear the adaptive hash index on all pages in the buffer pool. */
ATTRIBUTE_COLD void buf_pool_t::clear_hash_index() noexcept
{
std::set<dict_index_t*> garbage;
In the end, the entire adaptive hash index will be removed. */
ut_ad(s >= buf_page_t::UNFIXED || s == buf_page_t::REMOVE_HASH); # ifdefined UNIV_AHI_DEBUG || defined UNIV_DEBUG
block->n_pointers= 0; # endif /* UNIV_AHI_DEBUG || UNIV_DEBUG */ if (index->freed())
garbage.insert(index); else
index->search_info.ref_count= 0;
block->index= nullptr;
}
mysql_mutex_unlock(&mutex);
for (dict_index_t *index : garbage)
btr_search_lazy_free(index);
} #endif/* BTR_CUR_HASH_ADAPT */
#ifdef UNIV_DEBUG /** Check that all blocks are in a replaceable state. @returnaddressofanon-freeblock
@retval nullptr if all freed */ void buf_pool_t::assert_all_freed() noexcept
{
mysql_mutex_assert_owner(&mutex);
default: if (srv_read_only_mode)
{ /* The page cleaner is disabled in read-only mode. No pages
can be dirtied, so all of them must be clean. */
ut_ad(lsn == recv_sys.lsn ||
srv_force_recovery == SRV_FORCE_NO_LOG_REDO); break;
}
goto fixed_or_dirty;
}
if (!block->page.can_relocate())
fixed_or_dirty:
ib::fatal() << "Page " << block->page.id() << " still fixed or dirty";
}
} #endif/* UNIV_DEBUG */
/** Refresh the statistics used to print per-second averages. */ void buf_refresh_io_stats() noexcept
{
buf_pool.last_printout_time = time(NULL);
buf_pool.old_stat = buf_pool.stat;
}
for (ulint i = 0; i < n_blocks; i++) { const buf_block_t* block = get_nth_page(i);
ut_ad(block->page.frame == block->frame_address());
switch (constauto f = block->page.state()) { case buf_page_t::NOT_USED:
ut_ad(!block->page.in_LRU_list);
n_free++; break; case buf_page_t::MEMORY: case buf_page_t::REMOVE_HASH: /* do nothing */ break; default: if (f >= buf_page_t::READ_FIX
&& f < buf_page_t::WRITE_FIX) { /* A read-fixed block is not
necessarily in the page_hash yet. */ break;
}
ut_ad(f >= buf_page_t::FREED); const page_id_t id{block->page.id()};
ut_ad(page_hash.get(
id,
page_hash.cell_get(id.fold()))
== &block->page);
n_lru++;
}
}
/* Check dirty blocks. */
mysql_mutex_lock(&flush_list_mutex); for (buf_page_t* b = UT_LIST_GET_FIRST(flush_list); b;
b = UT_LIST_GET_NEXT(list, b)) {
ut_ad(b->in_file());
ut_ad(b->oldest_modification());
ut_ad(!fsp_is_system_temporary(b->id().space()));
n_flushing++;
for (i = 0; i < n_found; i++) {
index = dict_index_get_if_in_cache(index_ids[i]);
if (!index) {
ib::info() << "Block count for index "
<< index_ids[i] << " in buffer is about "
<< counts[i];
} else {
ib::info() << "Block count for index " << index_ids[i]
<< " in buffer is about " << counts[i]
<< ", index " << index->name
<< " of table " << index->table->name;
}
}
#ifdef UNIV_DEBUG /** @return the number of latched pages in the buffer pool */
ulint buf_get_latched_pages_number() noexcept
{
ulint fixed_pages_number= 0;
mysql_mutex_assert_owner(&buf_pool.mutex);
for (buf_page_t *b= UT_LIST_GET_FIRST(buf_pool.LRU); b;
b= UT_LIST_GET_NEXT(LRU, b)) if (b->state() > buf_page_t::UNFIXED)
fixed_pages_number++;
/*********************************************************************//**
Prints info of the buffer i/o. */ static void
buf_print_io_instance( /*==================*/
buf_pool_info_t*pool_info, /*!< in: buffer pool info */
FILE* file) /*!< in/out: buffer where to print */
{
ut_ad(pool_info);
/* Print some values to help us with visualizing what is
happening with LRU eviction. */
fprintf(file, "LRU len: " ULINTPF ", unzip_LRU len: " ULINTPF "\n" "I/O sum[" ULINTPF "]:cur[" ULINTPF "], " "unzip sum[" ULINTPF "]:cur[" ULINTPF "]\n",
pool_info->lru_len, pool_info->unzip_lru_len,
pool_info->io_sum, pool_info->io_cur,
pool_info->unzip_sum, pool_info->unzip_cur);
}
/*********************************************************************//**
Prints info of the buffer i/o. */ void
buf_print_io( /*=========*/
FILE* file) /*!< in/out: buffer where to print */
{
buf_pool_info_t pool_info;
/** Verify that post encryption checksum match with the calculated checksum. Thisfunctionshouldbecalledonlyiftablespacecontainscryptdatametadata. @parampagepageframe @paramfsp_flagscontentsofFSP_SPACE_FLAGS
@return whether the page is encrypted and valid */ bool buf_page_verify_crypt_checksum(const byte *page, uint32_t fsp_flags) noexcept
{ if (!fil_space_t::full_crc32(fsp_flags)) { return fil_space_verify_crypt_checksum(
page, fil_space_t::zip_size(fsp_flags));
}
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.180Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.