/** @name Modes for buf_page_get_gen */ /* @{ */ #define BUF_GET 10/*!< get always */ #define BUF_GET_RECOVER 9/*!< like BUF_GET, but in recv_sys.recover() */ #define BUF_GET_IF_IN_POOL 11/*!< get if in pool */ #define BUF_PEEK_IF_IN_POOL 12/*!< get if in pool, do not make
the block young in the LRU list */ #define BUF_GET_POSSIBLY_FREED 16 /*!< Like BUF_GET, but do not mind
if the file page has been freed. */ /* @} */
/** If LRU list of a buf_pool is less than this size then LRU eviction shouldnothappen.ThisisbecausewhenwedoLRUflushingwealsoput theblocksonfreelist.IfLRUlistisverysmallthenwecanendup
in thrashing. */ #define BUF_LRU_MIN_LEN 256
/** This structure defines information we will fetch from each buffer pool. It
will be used to print table IO stats */ struct buf_pool_info_t
{ /* General buffer pool info */
ulint pool_size; /*!< Buffer Pool size in pages */
ulint lru_len; /*!< Length of buf_pool.LRU */
ulint old_lru_len; /*!< buf_pool.LRU_old_len */
ulint free_list_len; /*!< free + lazy_allocate_size() */
ulint flush_list_len; /*!< Length of buf_pool.flush_list */
ulint n_pend_unzip; /*!< buf_pool.n_pend_unzip, pages
pending decompress */
ulint n_pend_reads; /*!< os_aio_pending_reads() */
ulint n_pending_flush_lru; /*!< Pages pending flush in LRU */
ulint n_pending_flush_list; /*!< Pages pending flush in FLUSH
LIST */
ulint n_pages_made_young; /*!< number of pages made young */
ulint n_pages_not_made_young; /*!< number of pages not made young */
ulint n_pages_read; /*!< buf_pool.n_pages_read */
ulint n_pages_created; /*!< buf_pool.n_pages_created */
ulint n_pages_written; /*!< buf_pool.n_pages_written */
ulint n_page_gets; /*!< buf_pool.n_page_gets */
ulint n_ra_pages_read_rnd; /*!< buf_pool.n_ra_pages_read_rnd,
number of pages readahead */
ulint n_ra_pages_read; /*!< buf_pool.n_ra_pages_read, number
of pages readahead */
ulint n_ra_pages_evicted; /*!< buf_pool.n_ra_pages_evicted, numberofreadaheadpagesevicted
without access */
ulint n_page_get_delta; /*!< num of buffer pool page gets since
last printout */
/* Buffer pool access stats */ double page_made_young_rate; /*!< page made young rate in pages
per second */ double page_not_made_young_rate;/*!< page not made young rate
in pages per second */ double pages_read_rate; /*!< num of pages read per second */ double pages_created_rate; /*!< num of pages create per second */ double pages_written_rate; /*!< num of pages written per second */
ulint page_read_delta; /*!< num of pages read since last
printout */
ulint young_making_delta; /*!< num of pages made young since
last printout */
ulint not_young_making_delta; /*!< num of pages not make young since
last printout */
/* Statistics about read ahead algorithm. */ double pages_readahead_rnd_rate;/*!< random readahead rate in pages per
second */ double pages_readahead_rate; /*!< readahead rate in pages per
second */ double pages_evicted_rate; /*!< rate of readahead page evicted
without access, in pages per second */
/* Stats about LRU eviction */
ulint unzip_lru_len; /*!< length of buf_pool.unzip_LRU
list */ /* Counters for LRU policy */
ulint io_sum; /*!< buf_LRU_stat_sum.io */
ulint io_cur; /*!< buf_LRU_stat_cur.io, num of IO
for current interval */
ulint unzip_sum; /*!< buf_LRU_stat_sum.unzip */
ulint unzip_cur; /*!< buf_LRU_stat_cur.unzip, num pagesdecompressedincurrent
interval */
}; #endif/* !UNIV_INNOCHECKSUM */
/** Print the given page_id_t object. @param[in,out]outtheoutputstream @param[in]page_idthepage_id_tobjecttobeprinted
@return the output stream */
std::ostream& operator<<(
std::ostream& out, const page_id_t page_id);
/** Try to buffer-fix a page. @paramblockguessedblock @paramidexpectedblock->page.id() @returnblockifitwasbuffer-fixed
@retval nullptr if the block no longer is valid */
buf_block_t *buf_page_optimistic_fix(buf_block_t *block, page_id_t id) noexcept
MY_ATTRIBUTE((nonnull, warn_unused_result));
/** Try to acquire a page latch after buf_page_optimistic_fix(). @paramblockbuffer-fixedblock @paramrw_latchRW_S_LATCHorRW_X_LATCH @parammodify_clockexpectedvalueofblock->modify_clock @parammtrmini-transaction @returnblockifthelatchwasacquired
@retval nullptr if block->unfix() was called because it no longer is valid */
buf_block_t *buf_page_optimistic_get(buf_block_t *block,
rw_lock_type_t rw_latch,
uint64_t modify_clock,
mtr_t *mtr) noexcept
MY_ATTRIBUTE((nonnull, warn_unused_result));
/** Try to S-latch a page. Suitableforusingwhenholdingthelock_syslatches(asitavoidsdeadlock). @param[in]page_idpageidentifier @param[in,out]mtrmini-transaction @returntheblock
@retval nullptr if an S-latch cannot be granted immediately */
buf_block_t *buf_page_try_get(const page_id_t page_id, mtr_t *mtr) noexcept;
/** Get read access to a compressed page (usually of type FIL_PAGE_TYPE_ZBLOBorFIL_PAGE_TYPE_ZBLOB2). Thepagemustbereleasedwiths_unlock(). @parampage_idpageidentifier
@return pointer to the block, s-latched */
buf_page_t *buf_page_get_zip(const page_id_t page_id) noexcept;
/** Get access to a database page. Buffered redo log may be applied. @param[in]page_idpageid @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]rw_latchlatchmode @param[in]guessguessedblockorNULL @param[in]modeBUF_GET,BUF_GET_IF_IN_POOL, orBUF_PEEK_IF_IN_POOL @param[in,out]mtrmini-transaction @param[out]errDB_SUCCESSorerrorcode @returnpointertotheblock
@retval nullptr if the block is corrupted or unavailable */
buf_block_t*
buf_page_get_gen( const page_id_t page_id,
ulint zip_size,
rw_lock_type_t rw_latch,
buf_block_t* guess,
ulint mode,
mtr_t* mtr,
dberr_t* err = nullptr) noexcept;
/** Initialize a page in the buffer pool. The page is usually not read fromafileevenifitcannotbefoundinthebufferbuf_pool.Thisisone ofthefunctionswhichperformtoablockastatetransitionNOT_USED=>LRU (theotherisbuf_page_get_gen()). @param[in,out]spacespaceobject @param[in]offsetoffsetofthetablespace @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in,out]mtrmini-transaction @param[in,out]free_blockpre-allocatedbufferblock
@return pointer to the block, page bufferfixed */
buf_block_t*
buf_page_create(fil_space_t *space, uint32_t offset,
ulint zip_size, mtr_t *mtr, buf_block_t *free_block)
noexcept;
/** Initialize a page in buffer pool while initializing the deferredtablespace @paramspace_idspaceidentfier @paramzip_sizeROW_FORMAT=COMPRESSEDpagesizeor0 @parammtrmini-transaction @paramfree_blockpre-allocatedbufferblock
@return pointer to the block, page bufferfixed */
buf_block_t*
buf_page_create_deferred(uint32_t space_id, ulint zip_size, mtr_t *mtr,
buf_block_t *free_block) noexcept;
/** Mark the page status as FREED for the given tablespace and page number. @param[in,out]spacetablespace @param[in]pagepagenumber
@param[in,out] mtr mini-transaction */ void buf_page_free(fil_space_t *space, uint32_t page, mtr_t *mtr);
/** Determine if a block is still close enough to the MRU end of the LRU list meaningthatitisnotindangerofgettingevictedandalsoimplying thatithasbeenaccessedrecently. Notethatthisisforheuristicsonlyanddoesnotreservebufferpool mutex. @param[in]bpagebufferpoolpage
@return whether bpage is close to MRU end of LRU */ inlinebool buf_page_peek_if_young(const buf_page_t *bpage);
/** Determine if a block should be moved to the start of the LRU list if thereisdangerofdroppingfromthebufferpool. @param[in]bpagebufferpoolpage
@return true if bpage should be made younger */ inlinebool buf_page_peek_if_too_old(const buf_page_t *bpage);
/********************************************************************//**
Increments the modify clock of a frame by 1. The caller must (1) own the
buf_pool.mutex and block bufferfix count has to be zero, (2) or own an x-lock
on the block. */
UNIV_INLINE void
buf_block_modify_clock_inc( /*=======================*/
buf_block_t* block); /*!< in: block */
/** Increment the pages_accessed count. */ void buf_inc_get(trx_t *trx) noexcept;
/** Check if a buffer is all zeroes. @param[in]bufdatatocheck
@return whether the buffer is all zeroes */ bool buf_is_zeroes(st_::span<const byte> buf) noexcept;
/** Check if a page is corrupt. @paramcheck_lsnwhetherFIL_PAGE_LSNshouldbechecked @paramread_bufdatabasepage @paramfsp_flagscontentsofFIL_SPACE_FLAGS
@return whether the page is corrupted */
buf_page_is_corrupted_reason
buf_page_is_corrupted(bool check_lsn, const byte *read_buf, uint32_t fsp_flags)
noexcept MY_ATTRIBUTE((warn_unused_result));
/** Read the key version from the page. In full crc32 format, keyversionisstoredat{0-3th}bytes.Inotherformat,itis storedin26thposition. @param[in]read_bufdatabasepage @param[in]fsp_flagstablespaceflags
@return key version of the page. */ inline uint32_t buf_page_get_key_version(const byte* read_buf,
uint32_t fsp_flags) noexcept
{
static_assert(FIL_PAGE_FCRC32_KEY_VERSION == 0, "compatibility"); return fil_space_t::full_crc32(fsp_flags)
? mach_read_from_4(my_assume_aligned<4>(read_buf))
: mach_read_from_4(my_assume_aligned<2>
(read_buf + FIL_PAGE_FILE_FLUSH_LSN_OR_KEY_VERSION));
}
/** Read the compression info from the page. In full crc32 format, compressioninfoisatMSBofpagetype.Inotherformat,itis storedinpagetype. @param[in]read_bufdatabasepage @param[in]fsp_flagstablespaceflags
@return true if page is compressed. */ inlinebool buf_page_is_compressed(const byte* read_buf, uint32_t fsp_flags) noexcept
{
uint16_t page_type= fil_page_get_type(read_buf); return fil_space_t::full_crc32(fsp_flags)
? !!(page_type & 1U << FIL_PAGE_COMPRESS_FCRC32_MARKER)
: page_type == FIL_PAGE_PAGE_COMPRESSED;
}
/** Get the compressed or uncompressed size of a full_crc32 page. @param[in]bufpage_compressedoruncompressedpage @param[out]compwhetherthepagecouldbecompressed @param[out]crwhetherthepagecouldbecorrupted
@return the payload size in the file page */ inline uint buf_page_full_crc32_size(const byte *buf, bool *comp, bool *cr)
noexcept
{
uint t = fil_page_get_type(buf);
uint page_size = uint(srv_page_size);
if (!(t & 1U << FIL_PAGE_COMPRESS_FCRC32_MARKER)) { return page_size;
}
t &= ~(1U << FIL_PAGE_COMPRESS_FCRC32_MARKER);
t <<= 8;
#ifndef UNIV_INNOCHECKSUM /** Dump a page to stderr. @param[in]read_bufdatabasepage
@param[in] zip_size compressed page size, or 0 */ void buf_page_print(const byte* read_buf, ulint zip_size = 0) noexcept
ATTRIBUTE_COLD __attribute__((nonnull)); /** Decompress a ROW_FORMAT=COMPRESSED block. @paramblockbufferpage @paramcheckwhethertoverifythepagechecksum
@return true if successful */ bool buf_zip_decompress(buf_block_t *block, bool check) noexcept;
#ifdef UNIV_DEBUG /** @return the number of latched pages in the buffer pool */
ulint buf_get_latched_pages_number() noexcept; #endif/* UNIV_DEBUG */ /*********************************************************************//**
Prints info of the buffer i/o. */ void
buf_print_io( /*=========*/
FILE* file); /*!< in: file where to print */
/** Refresh the statistics used to print per-second averages. */ void buf_refresh_io_stats() noexcept;
/** Verify that post encryption checksum match with the calculated checksum. Thisfunctionshouldbecalledonlyiftablespacecontainscryptdatametadata. @parampagepageframe @paramfsp_flagscontentsofFSP_SPACE_FLAGS
@return whether the page is encrypted and valid */ bool buf_page_verify_crypt_checksum(const byte *page, uint32_t fsp_flags) noexcept;
/** Calculate a ROW_FORMAT=COMPRESSED page checksum and update the page. @param[in,out]pagepagetoupdate
@param[in] size compressed page size */ void buf_flush_update_zip_checksum(buf_frame_t* page, ulint size) noexcept;
/** @brief The temporary memory structure.
NOTE!Thedefinitionappearshereonlyforothermodulesofthis
directory (buf) to see it. Do not use from outside! */
class buf_tmp_buffer_t
{ /** whether this slot is reserved */
std::atomic<bool> reserved; public: /** For encryption, the data needs to be copied to a separate buffer beforeit'sencrypted&written.Thebufferblockitselfcanbereplaced
while a write of crypt_buf to file is in progress. */
byte *crypt_buf; /** buffer for fil_page_compress(), for flushing page_compressed pages */
byte *comp_buf; /** pointer to resulting buffer after encryption or compression;
not separately allocated memory */
byte *out_buf;
/** Acquire the slot
@return whether the slot was acquired */ bool acquire() noexcept
{ return !reserved.exchange(true, std::memory_order_relaxed);}
/** Allocate a buffer for encryption, decryption or decompression. */ void allocate() noexcept
{ if (!crypt_buf)
crypt_buf= static_cast<byte*>
(aligned_malloc(srv_page_size, srv_page_size));
}
};
/** The common buffer control block structure
for compressed and uncompressed frames */
class buf_pool_t;
class buf_page_t
{ friend buf_pool_t; friend buf_block_t;
/** @name General fields */ /* @{ */
public: // FIXME: fix fil_iterate() /** Page id. Protected by buf_pool.page_hash.lock_get() when
the page is in buf_pool.page_hash. */
page_id_t id_; union { /** for in_file(): buf_pool.page_hash link;
protected by buf_pool.page_hash.lock_get() */
buf_page_t *hash; /** for state()==MEMORY that are part of recv_sys.pages and protectedbyrecv_sys.mutex,orpartofbtr_sea::partition::table
and protected by btr_sea::partition::blocks_mutex */ struct { /** number of recv_sys.pages entries stored in the block */
uint16_t used_records; /** the offset of the next free record */
uint16_t free_offset;
};
}; private: /** log sequence number of the START of the log entry written of the oldestmodificationtothisblockwhichhasnotyetbeenwritten tothedatafile;
0ifnomodificationsarepending; 1ifnomodificationsarepending,buttheblockisinbuf_pool.flush_list; 2ifmodificationsarepending,buttheblockisnotinbuf_pool.flush_list
(because id().space() is the temporary tablespace). */
Atomic_relaxed<lsn_t> oldest_modification_;
public: /** state() of unused block (in buf_pool.free list) */ static constexpr uint32_t NOT_USED= 0; /** state() of block allocated as general-purpose memory */ static constexpr uint32_t MEMORY= 1; /** state() of block that is being freed */ static constexpr uint32_t REMOVE_HASH= 2; /** smallest state() of a buffer page that is freed in the tablespace */ static constexpr uint32_t FREED= 3; /* unused state: 1U<<29 */ /** smallest state() for a block that belongs to buf_pool.LRU */ static constexpr uint32_t UNFIXED= 2U << 29; /** smallest state() of a (re)initialized page (no doublewrite needed) */ static constexpr uint32_t REINIT= 3U << 29; /** smallest state() for an io-fixed block */ static constexpr uint32_t READ_FIX= 4U << 29; /* unused state: 5U<<29 */ /** smallest state() for a write-fixed block */ static constexpr uint32_t WRITE_FIX= 6U << 29; /** smallest state() for a write-fixed block (no doublewrite was used) */ static constexpr uint32_t WRITE_FIX_REINIT= 7U << 29; /** buf_pool.LRU status mask in state() */ static constexpr uint32_t LRU_MASK= 7U << 29;
/** lock covering the contents of frame() */
block_lock lock; /** pointer to aligned, uncompressed page frame of innodb_page_size */
byte *frame; /* @} */ /** ROW_FORMAT=COMPRESSED page; zip.data (but not the data it points to)
is also protected by buf_pool.mutex */
page_zip_des_t zip; #ifdef UNIV_DEBUG /** whether this->LRU is in buf_pool.LRU (in_file());
protected by buf_pool.mutex */ bool in_LRU_list; /** whether this is in buf_pool.page_hash (in_file());
protected by buf_pool.mutex */ bool in_page_hash; /** whether this->list is in buf_pool.free (state() == NOT_USED);
protected by buf_pool.flush_list_mutex */ bool in_free_list; #endif/* UNIV_DEBUG */ /** list member in one of the lists of buf_pool; protected by buf_pool.mutexorbuf_pool.flush_list_mutex
UT_LIST_NODE_T(buf_page_t) LRU; /*!< node of the LRU list */ unsigned old:1; /*!< TRUE if the block is in the old
blocks in buf_pool.LRU_old */ unsigned freed_page_clock:31;/*!< the value of buf_pool.freed_page_clock whenthisblockwasthelast timeputtotheheadofthe LRUlist;athreadisallowed toreadthisforheuristic purposeswithoutholdingany
mutex or latch */ /* @} */
Atomic_counter<unsigned> access_time; /*!< time of first access, or 0iftheblockwasneveraccessed
in the buffer pool. */
buf_page_t() : id_{0}
{
static_assert(NOT_USED == 0, "compatibility");
memset((void*) this, 0, sizeof *this);
}
uint32_t buf_fix_count() const { return buf_fix_count(state()); } /** Check if a file block is io-fixed. @paramsstate()
@return whether s corresponds to an io-fixed block */ staticbool is_io_fixed(uint32_t s) noexcept
{ ut_ad(s >= FREED); return s >= READ_FIX; } /** Check if a file block is read-fixed. @paramsstate()
@return whether s corresponds to a read-fixed block */ staticbool is_read_fixed(uint32_t s) noexcept
{ return is_io_fixed(s) && s < WRITE_FIX; } /** Check if a file block is write-fixed. @paramsstate()
@return whether s corresponds to a write-fixed block */ staticbool is_write_fixed(uint32_t s) noexcept
{ ut_ad(s >= FREED); return s >= WRITE_FIX; }
/** @return whether this block is read or write fixed; read_complete()orwrite_complete()willalwaysrelease
the io-fix before releasing U-lock or X-lock */ bool is_io_fixed() const noexcept { return is_io_fixed(state()); } /** @return whether this block is write fixed;
write_complete() will always release the write-fix before releasing U-lock */ bool is_write_fixed() const noexcept { return is_write_fixed(state()); } /** @return whether this block is read fixed */ bool is_read_fixed() const noexcept { return is_read_fixed(state()); }
/** @return if this belongs to buf_pool.unzip_LRU */ bool belongs_to_unzip_LRU() const noexcept
{ return UNIV_LIKELY_NULL(zip.data) && frame; }
/** @return the log sequence number of the oldest pending modification @retval0iftheblockisbeingremovedfrom(ornotin)buf_pool.flush_list @retval1iftheblockisinbuf_pool.flush_listbutnotmodified @retval2iftheblockbelongstothetemporarytablespaceand
has unwritten changes */
lsn_t oldest_modification() const noexcept { return oldest_modification_; } /** @return the log sequence number of the oldest pending modification, @retval0iftheblockisdefinitelynotinbuf_pool.flush_list @retval1iftheblockisinbuf_pool.flush_listbutnotmodified @retval2iftheblockbelongstothetemporarytablespaceand
has unwritten changes */
lsn_t oldest_modification_acquire() const noexcept
{ return oldest_modification_.load(std::memory_order_acquire); } /** Set oldest_modification when adding to buf_pool.flush_list */ inlinevoid set_oldest_modification(lsn_t lsn) noexcept; /** Clear oldest_modification after removing from buf_pool.flush_list */ inlinevoid clear_oldest_modification() noexcept; /** Reset the oldest_modification when marking a persistent page freed */ void reset_oldest_modification() noexcept
{
ut_ad(oldest_modification() > 2);
oldest_modification_.store(1, std::memory_order_release);
}
/** Complete a read of a page. @paramnodedatafile @paramrecoveryrecv_recovery_is_on() @returnwhethertheoperationsucceeded @retvalDB_SUCCESSifthereadsucceeded;callermustunfix() @retvalDB_PAGE_CORRUPTEDifthechecksumorthepageIDisincorrect
@retval DB_DECRYPTION_FAILED if the page cannot be decrypted */
dberr_t read_complete(const fil_node_t &node, bool recovery) noexcept;
/** Wait for read_complete() by invoking lock.s_lock_nospin().
@param trx transaction (for updating trx->active_handler_stats) */ void read_wait(trx_t *trx) noexcept;
/** Release a write fix after a page write was completed. @parampersistentwhetherthepagebelongstoapersistenttablespace @paramerrorwhetheranerrormayhaveoccurredwhilewriting
@param state recently read state() value with the correct io-fix */ void write_complete(bool persistent, bool error, uint32_t state) noexcept;
/** Write a flushable page to a file or free a freeable block. @paramspacetablespace
@return whether a page write was initiated and buf_pool.mutex released */ bool flush(fil_space_t *space) noexcept;
/** Notify that a page in a temporary tablespace has been modified. */ void set_temp_modified() noexcept
{
ut_ad(fsp_is_system_temporary(id().space()));
ut_ad(in_file());
ut_ad((oldest_modification() | 2) == 2);
oldest_modification_= 2;
}
/** Prepare to release a file page to buf_pool.free. */ void free_file_page() noexcept
{
assert((zip.fix.fetch_sub(REMOVE_HASH - MEMORY)) == REMOVE_HASH); /* buf_LRU_block_free_non_file_page() asserts !oldest_modification() */
ut_d(oldest_modification_= 0;)
id_= page_id_t(~0ULL);
}
/** @return the ROW_FORMAT=COMPRESSED physical size, in bytes
@retval 0 if not compressed */
ulint zip_size() const noexcept
{ return zip.ssize ? (UNIV_ZIP_SIZE_MIN >> 1) << zip.ssize : 0;
}
/** @return the byte offset of the page within a file */
os_offset_t physical_offset() const noexcept
{
os_offset_t o= id().page_no(); return zip.ssize
? o << (zip.ssize + (UNIV_ZIP_SIZE_SHIFT_MIN - 1))
: o << srv_page_size_shift;
}
/** @return whether the block is mapped to a data file */ bool in_file() const noexcept { return state() >= FREED; }
/** @return whether the block can be relocated in memory.
The block can be dirty, but it must not be I/O-fixed or bufferfixed. */ inlinebool can_relocate() const noexcept; /** @return whether the block has been flagged old in buf_pool.LRU */ inlinebool is_old() const noexcept; /** Set whether a block is old in buf_pool.LRU */ inlinevoid set_old(bool old) noexcept; /** Flag a page accessed in buf_pool
@return whether this is not the first access */ bool set_accessed() noexcept
{ if (is_accessed()) returntrue;
access_time= static_cast<uint32_t>(ut_time_ms()); returnfalse;
} /** @return ut_time_ms() at the time of first access of a block in buf_pool
@retval 0 if not accessed */ unsigned is_accessed() const noexcept
{ ut_ad(in_file()); return access_time; }
};
/** The buffer control block structure */
struct buf_block_t{
/** @name General fields */ /* @{ */
buf_page_t page; /*!< page information; this must bethefirstfield,sothat buf_pool.page_hashcanpoint
to buf_page_t or buf_block_t */ #ifdef UNIV_DEBUG /** whether unzip_LRU is in buf_pool.unzip_LRU (in_file()&&frame&&zip.data);
protected by buf_pool.mutex */ bool in_unzip_LRU_list; #endif /** member of buf_pool.unzip_LRU (if belongs_to_unzip_LRU()) */
UT_LIST_NODE_T(buf_block_t) unzip_LRU; /* @} */ /** @name Optimistic search field */ /* @{ */
ib_uint64_t modify_clock; /*!< this clock is incremented every timeapointertoarecordonthe pagemaybecomeobsolete;thisis usedintheoptimisticcursor positioning:ifthemodifyclockhas notchanged,weknowthatthepointer isstillvalid;thisfieldmaybe changedifthethread(1)ownsthe poolmutexandthepageisnot bufferfixed,or(2)thethreadhasan
x-latch on the block */ /* @} */ #ifdef BTR_CUR_HASH_ADAPT /** @name Hash search fields */ /* @{ */ /** flag: (true=first, false=last) identical-prefix key is included */ static constexpr uint32_t LEFT_SIDE= 1U << 31;
/** counter which controls building of a new hash index for the page;
may be nonzero even if !index */
Atomic_relaxed<uint16_t> n_hash_helps; # ifdefined UNIV_AHI_DEBUG || defined UNIV_DEBUG /** number of pointers from the btr_sea::partition::table;
!index implies n_pointers == 0 */
Atomic_counter<uint16_t> n_pointers; # define assert_block_ahi_empty(block) ut_a(!(block)->n_pointers) # define assert_block_ahi_valid(b) ut_a((b)->index || !(b)->n_pointers) # else/* UNIV_AHI_DEBUG || UNIV_DEBUG */ # define assert_block_ahi_empty(block) /* nothing */ # define assert_block_ahi_valid(block) /* nothing */ # endif /* UNIV_AHI_DEBUG || UNIV_DEBUG */ /** index for which the adaptive hash index has been created, ornullptrifthepagedoesnotexistintheindex.
May be modified while holding exclusive btr_sea::partition::latch. */
Atomic_relaxed<dict_index_t*> index; /* @} */ #else/* BTR_CUR_HASH_ADAPT */ # define assert_block_ahi_empty(block) /* nothing */ # define assert_block_ahi_valid(block) /* nothing */ #endif/* BTR_CUR_HASH_ADAPT */ void fix() noexcept { page.fix(); }
uint32_t unfix() noexcept { return page.unfix(); }
/** @return the physical size, in bytes */
ulint physical_size() const noexcept { return page.physical_size(); }
/** @return the ROW_FORMAT=COMPRESSED physical size, in bytes
@retval 0 if not compressed */
ulint zip_size() const noexcept { return page.zip_size(); }
/** Initialize the block. @parampage_idpageidentifier @paramzip_sizeROW_FORMAT=COMPRESSEDpagesize,or0
@param state initial state() */ void initialise(const page_id_t page_id, ulint zip_size, uint32_t state)
noexcept;
/** A "Hazard Pointer" class used to iterate over buf_pool.LRU or buf_pool.flush_list.Ahazardpointerisabuf_page_tpointer whichweintendtoiterateovernextandwewantitremainvalid
even after we release the mutex that protects the list. */ class HazardPointer
{ public: virtual ~HazardPointer() = default;
/** @return current value */
buf_page_t *get() const noexcept
{ mysql_mutex_assert_owner(m_mutex); return m_hp; }
/** Set current value
@param bpage buffer block to be set as hp */ void set(buf_page_t *bpage) noexcept
{
mysql_mutex_assert_owner(m_mutex);
ut_ad(!bpage || bpage->in_file());
m_hp= bpage;
}
/** Checks if a bpage is the hp @parambpagebufferblocktobecompared
@return true if it is hp */ bool is_hp(const buf_page_t *bpage) const noexcept
{ mysql_mutex_assert_owner(m_mutex); return bpage == m_hp; }
/** Adjust the value of hp. This happens when some otherthreadworkingonthesamelistattemptsto
remove the hp from the list. */ virtualvoid adjust(const buf_page_t*) noexcept = 0;
#ifdef UNIV_DEBUG /** mutex that protects access to the m_hp. */ const mysql_mutex_t *m_mutex= nullptr; #endif/* UNIV_DEBUG */
/** Class implementing buf_pool.flush_list hazard pointer */ class FlushHp : public HazardPointer
{ public:
~FlushHp() override = default;
/** Adjust the value of hp. This happens when some otherthreadworkingonthesamelistattemptsto removethehpfromthelist.
@param bpage buffer block to be compared */
MY_ATTRIBUTE((nonnull)) void adjust(const buf_page_t *bpage) noexcept override
{ /* We only support reverse traversal for now. */ if (is_hp(bpage))
m_hp= UT_LIST_GET_PREV(list, m_hp);
ut_ad(!m_hp || m_hp->oldest_modification());
}
};
/** Class implementing buf_pool.LRU hazard pointer */ class LRUHp : public HazardPointer { public:
~LRUHp() override = default;
/** Adjust the value of hp. This happens when some otherthreadworkingonthesamelistattemptsto removethehpfromthelist.
@param bpage buffer block to be compared */
MY_ATTRIBUTE((nonnull)) void adjust(const buf_page_t *bpage) noexcept override
{ /** We only support reverse traversal for now. */ if (is_hp(bpage))
m_hp= UT_LIST_GET_PREV(LRU, m_hp);
ut_ad(!m_hp || m_hp->in_LRU_list);
}
};
/** Special purpose iterators to be used when scanning the LRU list. Theideaisthatwhenonethreadfinishesthescanitleavesthe itrinthatpositionandtheotherthreadcanstartscanfrom
there */ class LRUItr : public LRUHp { public:
~LRUItr() override = default;
/** Select from where to start a scan. If we have scanned toodeepintotheLRUlistitresetsthevaluetothetail oftheLRUlist.
@return buf_page_t from where to start scan. */ inline buf_page_t *start() noexcept;
};
/** Struct that is embedded in the free zip blocks */ struct buf_buddy_free_t { union {
ulint size; /*!< size of the block */
byte bytes[FIL_PAGE_DATA]; /*!< stamp[FIL_PAGE_ARCH_LOG_NO_OR_SPACE_ID] ==BUF_BUDDY_FREE_STAMPdenotesafree block.Ifthespace_idfieldofbuddy block!=BUF_BUDDY_FREE_STAMP,theblock isnotinanyzip_freelist.Ifthe space_idisBUF_BUDDY_FREE_STAMPthen stamp[0]willcontainthe
buddy block size. */
} stamp;
buf_page_t bpage; /*!< Embedded bpage descriptor */
UT_LIST_NODE_T(buf_buddy_free_t) list; /*!< Node of zip_free list */
};
/** @brief The buffer pool statistics structure;
protected by buf_pool.mutex unless otherwise noted. */ struct buf_pool_stat_t{ /** Initialize the counters */ void init() noexcept { memset((void*) this, 0, sizeof *this); }
/** number of pages accessed; aggregates trx_t::pages_accessed */ union {
Atomic_counter<ulint> n_page_gets{0};
ulint n_page_gets_nonatomic;
};
ulint n_pages_read; /*!< number read operations */
ulint n_pages_written;/*!< number write operations */
ulint n_pages_created;/*!< number of pages created
in the pool with no read */
ulint n_ra_pages_read_rnd;/*!< number of pages read in
as part of random read ahead */
ulint n_ra_pages_read;/*!< number of pages read in
as part of read ahead */
ulint n_ra_pages_evicted;/*!< number of read ahead pagesthatareevictedwithout
being accessed */
ulint n_pages_made_young; /*!< number of pages made young, in
buf_page_make_young() */
ulint n_pages_not_made_young; /*!< number of pages not made youngbecausethefirstaccess wasnotlongenoughago,in
buf_page_peek_if_too_old() */ /** number of waits for eviction */
ulint LRU_waits;
ulint LRU_bytes; /*!< LRU size in bytes */
};
/** Statistics of buddy blocks of a given size. */ struct buf_buddy_stat_t { /** Number of blocks allocated from the buddy system. */
ulint used; /** Number of blocks relocated by the buddy system. */
ib_uint64_t relocated; /** Total duration of block relocations, in microseconds. */
ib_uint64_t relocated_usec;
};
/** The buffer pool */ class buf_pool_t
{ /** arrays of buf_block_t followed by page frames; aligedtoandrepeatingeveryinnodb_buffer_pool_extent_size;
each extent comprises pages_in_extent[] blocks */
alignas(CPU_LEVEL1_DCACHE_LINESIZE) char *memory; /** the allocation of the above memory, possibly including some
alignment loss at the beginning */ char *memory_unaligned; /** the virtual address range size of memory_unaligned */
size_t size_unaligned; #ifdef UNIV_PFS_MEMORY /** the "owner thread" of the buffer pool allocation */
PSI_thread *owner; #endif /** initialized number of block descriptors */
size_t n_blocks; /** number of blocks that need to be freed in shrink() */
size_t n_blocks_to_withdraw; /** first block to withdraw in shrink() */ const buf_page_t *first_to_withdraw;
/** amount of memory allocated to the buffer pool and descriptors;
protected by mutex */
Atomic_relaxed<size_t> size_in_bytes;
public: /** The requested innodb_buffer_pool_size */
size_t size_in_bytes_requested; #ifdefined __linux__ || !defined DBUG_OFF /** The minimum allowed innodb_buffer_pool_size in garbage_collect() */
size_t size_in_bytes_auto_min; #endif #if SIZEOF_SIZE_T < 8 || defined _AIX || defined HAVE_valgrind /* In constrained environments, innodb_buffer_pool_size_max willdefaulttotheinitialinnodb_buffer_pool_size,thatis, bydefault,itwillnotbepossibletoincreaseinnodb_buffer_pool_size.
InMemorySanitizerandpossiblyValgrindmemcheck,anyvirtualmemory allocationwouldbebackedbyoneormorecopiesofshadowbitsofthe samesizethatcouldbeallocatedandinitializedevenfordummy mappingscreatedbymmap(2)withPROT_NONE.Wedonotwantsignificant
overhead beyond the actual innodb_buffer_pool_size. */ static constexpr size_t size_in_bytes_max_default{0},
size_in_bytes_max_minimum{0}; #else static constexpr size_t size_in_bytes_max_default{8ULL << 40},
size_in_bytes_max_minimum{innodb_buffer_pool_extent_size}; #endif /** The maximum allowed innodb_buffer_pool_size */
size_t size_in_bytes_max;
/** @return the current size of the buffer pool, in bytes */
size_t curr_pool_size() const noexcept { return size_in_bytes; }
/** @return the current size of the buffer pool, in pages */
TPOOL_SUPPRESS_TSAN size_t curr_size() const noexcept { return n_blocks; } /** @return the maximum usable size of the buffer pool, in pages */
TPOOL_SUPPRESS_TSAN size_t usable_size() const noexcept
{ return n_blocks - n_blocks_to_withdraw - UT_LIST_GET_LEN(withdrawn); }
/** Determine the used size of the buffer pool in bytes. @paramn_blockssizeofthebufferpoolinblocks
@return the size needed for n_blocks in bytes, for innodb_page_size */ static size_t blocks_in_bytes(size_t n_blocks) noexcept;
#ifdefined(DBUG_OFF) && defined(HAVE_MADVISE) && defined(MADV_DODUMP) /** Enable buffers to be dumped to core files.
@return number of errors found in madvise() calls */ staticint madvise_do_dump() noexcept; #endif
#ifdefined __linux__ || defined __FreeBSD__ /** Include or exclude the buffer pool from core dump. */ void core_advise() noexcept
{
mysql_mutex_assert_owner(&mutex);
madvise(memory, size_in_bytes, in_core_dump ? MADV_DODUMP : MADV_DONTDUMP);
} #endif
/** Hash cell chain in page_hash_table */ struct hash_chain
{ /** pointer to the first block */
buf_page_t *first;
}; private: /** Determine the number of blocks in a buffer pool of a particular size. @paramsize_in_bytesinnodb_buffer_pool_sizeinbytes
@return number of buffer pool pages */ static size_t get_n_blocks(size_t size_in_bytes) noexcept;
/** The outcome of shrink() */ enum shrink_status{SHRINK_DONE= -1, SHRINK_IN_PROGRESS= 0, SHRINK_ABORT};
/** Attempt to shrink the buffer pool. @paramsizerequestedinnodb_buffer_pool_sizeinbytes
@retval whether the shrinking was completed */
ATTRIBUTE_COLD shrink_status shrink(size_t size) noexcept;
/** Finish shrinking the buffer pool. @paramsizethenewinnodb_buffer_pool_sizeinbytes
@param reduced how much the innodb_buffer_pool_size was reduced */ inlinevoid shrunk(size_t size, size_t reduced) noexcept;
/** Create the buffer pool.
@return whether the creation failed */ bool create() noexcept;
/** Clean up after successful create() */ void close() noexcept;
/** Resize the buffer pool. @paramsizerequestedinnodb_buffer_pool_sizeinbytes
@param trx current connnection */
ATTRIBUTE_COLD void resize(size_t size, THD *thd) noexcept;
/** Collect garbage (release pages from the LRU list) */ inlinevoid garbage_collect() noexcept;
/** Determine whether a frame needs to be withdrawn during resize(). @paramptrpointerwithinabuf_page_t::frame @paramsizesize_in_bytes_requested
@return whether the frame will be withdrawn */ bool will_be_withdrawn(const byte *ptr, size_t size) const noexcept
{ constchar *p= reinterpret_cast<constchar*>(ptr);
ut_ad(!p || p >= memory);
ut_ad(p < memory + size_in_bytes_max); return p >= memory + size;
}
/** Withdraw a block if needed in case resize() is shrinking. @parambpagebufferpoolblock
@return whether the block was withdrawn */
ATTRIBUTE_COLD bool withdraw(buf_page_t &bpage) noexcept;
/** Release and evict a corrupted page. @parambpagex-latchedpagethatwasfoundcorrupted
@param state expected current state of the page */
ATTRIBUTE_COLD void corrupted_evict(buf_page_t *bpage, uint32_t state)
noexcept;
/** Release a memory block to the buffer pool. */
ATTRIBUTE_COLD void free_block(buf_block_t *block) noexcept;
#ifdef UNIV_DEBUG /** Find a block that points to a ROW_FORMAT=COMPRESSED page @paramdatapointertothestartofaROW_FORMAT=COMPRESSEDpageframe @paramshiftnumberofleastsignificantaddressbitstoignore @returntheblock
@retval nullptr if not found */ const buf_block_t *contains_zip(constvoid *data, size_t shift= 0) const noexcept; /** Assert that all buffer pool pages are in a replaceable state */ void assert_all_freed() noexcept; #endif/* UNIV_DEBUG */
#ifdef BTR_CUR_HASH_ADAPT /** Clear the adaptive hash index on all pages in the buffer pool. */ void clear_hash_index() noexcept; #endif/* BTR_CUR_HASH_ADAPT */
/** @returnthesmallestoldest_modificationlsnforanypage
@retval empty_lsn if all modified persistent pages have been flushed */
lsn_t get_oldest_modification(lsn_t empty_lsn) noexcept
{
mysql_mutex_assert_owner(&flush_list_mutex); while (buf_page_t *bpage= UT_LIST_GET_LAST(flush_list))
{
ut_ad(!fsp_is_system_temporary(bpage->id().space()));
lsn_t lsn= bpage->oldest_modification(); if (lsn != 1)
{
ut_ad(lsn > 2); return lsn;
}
delete_from_flush_list(bpage);
} return empty_lsn;
}
/** Look up the block descriptor for a page frame address. @paramptraddresswithinavalidpageframe
@return the corresponding block descriptor */ static buf_block_t *block_from(constvoid *ptr) noexcept;
/** Access a block while holding the buffer pool mutex. @parampospositionbetween0andget_n_pages()
@return the block descriptor */
buf_block_t *get_nth_page(size_t pos) const noexcept;
#ifdef UNIV_DEBUG /** Determine if an object is within the curr_pool_size() andassociatedwithanuncompressedpage. @paramptrmemoryobject(notdereferenced)
@return whether the object is valid in the current buffer pool */ bool is_uncompressed_current(constvoid *ptr) const noexcept
{ const ptrdiff_t d= static_cast<constchar*>(ptr) - memory; return d >= 0 && size_t(d) < curr_pool_size();
} #endif
public: /** page_fix() mode of operation */ enum page_fix_conflicts{ /** Fetch if in the buffer pool, also blocks marked as free */
FIX_ALSO_FREED= -1, /** Fetch, waiting for page read completion */
FIX_WAIT_READ, /** Fetch, but avoid any waits for */
FIX_NOWAIT
};
/** Look up and buffer-fix a page. Note:Ifthepageisread-fixed(beingreadintothebufferpool), wewouldhavetowaitforthepagelatchbeforedeterminingifthepage isaccessible(itcouldbecorruptedandhavebeenevictedagain). Ifthecallerisholdingotherpagelatchessothatwaitingforthis pagelatchcouldleadtolockorderinversion(latchingorderviolation), themodec=FIX_WAIT_READmustnotbeused. @paramidpageidentifier @paramerrerrorcode(willonlybeassignedwhenreturningnullptr) @paramtrxtransactionattachedtocurrentconnection @paramchowtohandleconflicts @returnundologpage,buffer-fixed @retval-1ifc=FIX_NOWAITandbuffer-fixingwouldrequirewaiting
@retval nullptr if the undo page was corrupted or freed */
buf_block_t *page_fix(const page_id_t id, dberr_t *err, trx_t *trx,
page_fix_conflicts c) noexcept;
/** Validate a block descriptor. @parambblockdescriptorthatmaybeinvalidaftershrink() @paramlatchpage_hashlatchforid @paramidpageidentifier @returnb->page.fix()ifb->page.id()==id
@retval 0 if b is invalid */
uint32_t page_guess(buf_block_t *b, page_hash_latch &latch, const page_id_t id) noexcept;
/** Decompress a page and relocate the block descriptor @parambbuffer-fixedcompressed-onlyROW_FORMAT=COMPRESSEDpage @paramchainhashtablechainforb->id().fold() @returnthedecompressedblock,x-latchedandread-fixed
@retval nullptr if the decompression failed (b->unfix() will be invoked) */
ATTRIBUTE_COLD __attribute__((nonnull, warn_unused_result))
buf_block_t *unzip(buf_page_t *b, hash_chain &chain) noexcept;
/** @return whether the buffer pool contains a page @parampage_idpageidentifier
@param chain hash table chain for page_id.fold() */
TRANSACTIONAL_TARGET bool page_hash_contains(const page_id_t page_id, hash_chain &chain) noexcept;
/** @return whether less than 1/4 of the buffer pool is available */ bool running_out() const noexcept;
/** @return whether the buffer pool is running low */ bool need_LRU_eviction() const noexcept;
/** @return number of blocks resize() needs to evict from the buffer pool */
size_t is_shrinking() const noexcept
{
mysql_mutex_assert_owner(&mutex); return n_blocks_to_withdraw + UT_LIST_GET_LEN(withdrawn);
}
/** @return number of blocks in resize() waiting to be withdrawn */
size_t to_withdraw() const noexcept
{
mysql_mutex_assert_owner(&mutex); return n_blocks_to_withdraw;
}
/** @return the shrinking size of the buffer pool, in bytes
@retval 0 if resize() is not shrinking the buffer pool */
size_t shrinking_size() const noexcept
{ return is_shrinking() ? size_in_bytes_requested : 0; }
#ifdef UNIV_DEBUG /** Validate the buffer pool. */ void validate() noexcept; #endif/* UNIV_DEBUG */ #ifdefined UNIV_DEBUG_PRINT || defined UNIV_DEBUG /** Write information of the buf_pool to the error log. */ void print() noexcept; #endif/* UNIV_DEBUG_PRINT || UNIV_DEBUG */
/** Remove a block from the LRU list.
@return the predecessor in the LRU list */
buf_page_t *LRU_remove(buf_page_t *bpage) noexcept
{
mysql_mutex_assert_owner(&mutex);
ut_ad(bpage->in_LRU_list);
ut_ad(bpage->in_page_hash);
ut_ad(bpage->in_file());
lru_hp.adjust(bpage);
lru_scan_itr.adjust(bpage);
ut_d(bpage->in_LRU_list= false);
buf_page_t *prev= UT_LIST_GET_PREV(LRU, bpage);
UT_LIST_REMOVE(LRU, bpage); return prev;
}
/** Number of pages to read ahead */ static constexpr uint32_t READ_AHEAD_PAGES= 64;
/** current statistics; protected by mutex */
buf_pool_stat_t stat; /** old statistics; protected by mutex */
buf_pool_stat_t old_stat;
/** @name General fields */ /* @{ */
ulint LRU_old_ratio; /*!< Reserve this much of the buffer
pool for "old" blocks */ /** read-ahead request size in pages */
Atomic_counter<uint32_t> read_ahead_area;
/** Hash table with singly-linked overflow lists */ struct page_hash_table
{
static_assert(CPU_LEVEL1_DCACHE_LINESIZE >= 64, "less than 64 bytes");
static_assert(!(CPU_LEVEL1_DCACHE_LINESIZE & 63), "not a multiple of 64 bytes");
/** Number of array[] elements per page_hash_latch.
Must be one less than a power of 2. */ static constexpr size_t ELEMENTS_PER_LATCH= 64 / sizeof(void*) - 1; static constexpr size_t EMPTY_SLOTS_PER_LATCH=
((CPU_LEVEL1_DCACHE_LINESIZE / 64) - 1) * (64 / sizeof(void*));
/** number of payload elements in array[] */
Atomic_relaxed<ulint> n_cells; /** the hash table, with pad(n_cells) elements, aligned to L1 cache size */
hash_chain *array;
/** Create the hash table.
@param n the lower bound of n_cells */ void create(ulint n) noexcept;
/** @return the index of an array element */
ulint calc_hash(ulint fold) const noexcept
{ return calc_hash(fold, n_cells); } /** @return raw array index converted to padded index */ static ulint pad(ulint h) noexcept
{
ulint latches= h / ELEMENTS_PER_LATCH;
ulint empty_slots= latches * EMPTY_SLOTS_PER_LATCH; return1 + latches + empty_slots + h;
} private: /** @return the index of an array element */ static ulint calc_hash(ulint fold, ulint n_cells) noexcept
{ return pad(fold % n_cells);
} public: /** @return the latch covering a hash table chain */ static page_hash_latch &lock_get(hash_chain &chain) noexcept
{
static_assert(!((ELEMENTS_PER_LATCH + 1) & ELEMENTS_PER_LATCH), "must be one less than a power of 2"); const size_t addr= reinterpret_cast<size_t>(&chain);
ut_ad(addr & (ELEMENTS_PER_LATCH * sizeof chain)); return *reinterpret_cast<page_hash_latch*>
(addr & ~(ELEMENTS_PER_LATCH * sizeof chain));
}
/** Get a hash table slot. */
hash_chain &cell_get(ulint fold) const
{ return array[calc_hash(fold, n_cells)]; }
/** Append a block descriptor to a hash bucket chain. */ void append(hash_chain &chain, buf_page_t *bpage) noexcept;
/** Remove a block descriptor from a hash bucket chain. */ inlinevoid remove(hash_chain &chain, buf_page_t *bpage) noexcept; /** Replace a block descriptor with another. */ inlinevoid replace(hash_chain &chain, buf_page_t *old, buf_page_t *bpage)
noexcept;
/** Look up a page in a hash bucket chain. */ inline buf_page_t *get(const page_id_t id, const hash_chain &chain) const
noexcept;
};
/** Buffer pool mutex */
alignas(CPU_LEVEL1_DCACHE_LINESIZE) mysql_mutex_t mutex;
/** innodb_lru_scan_depth; number of blocks scanned in LRU flush batch;
protected by buf_pool_t::mutex */
ulong LRU_scan_depth; /** innodb_flush_neighbors; whether or not to flush neighbors of a block;
protected by buf_pool_t::mutex */
ulong flush_neighbors;
/** Hash table of file pages (buf_page_t::in_file() holds),
indexed by page_id_t. Protected by both mutex and page_hash.lock_get(). */
page_hash_table page_hash;
/** number of pending unzip() */
Atomic_counter<ulint> n_pend_unzip;
time_t last_printout_time; /*!< when buf_print_io was last time
called */
buf_buddy_stat_t buddy_stat[BUF_BUDDY_SIZES_MAX + 1]; /*!< Statistics of buddy system,
indexed by block size */
/* @} */
/** number of index page splits */
Atomic_counter<ulint> pages_split;
/** mutex protecting flush_list, buf_page_t::set_oldest_modification()
and buf_page_t::list pointers when !oldest_modification() */
alignas(CPU_LEVEL1_DCACHE_LINESIZE) mysql_mutex_t flush_list_mutex; /** "hazard pointer" for flush_list scans; protected by flush_list_mutex */
FlushHp flush_hp; /** flush_list size in bytes; protected by flush_list_mutex */
ulint flush_list_bytes; /** possibly modified persistent pages (a subset of LRU);
os_aio_pending_writes() is approximately COUNT(is_write_fixed()) */
UT_LIST_BASE_NODE_T(buf_page_t) flush_list; /** number of blocks ever added to flush_list;
sometimes protected by flush_list_mutex */
size_t flush_list_requests;
/** Number of pending LRU flush * LRU_FLUSH + FLUSH_LIST_ACTIVE flag.
Protected by flush_list_mutex. */ unsigned page_cleaner_status;
/** Whether the page cleaner is sleeping due to being idle. Writesoccurunderflush_list_mutex;readsarelock-free(usedby
the early-return in buf_flush_ahead() on a busy cleaner). */
Atomic_relaxed<bool> page_cleaner_idle_flag;
/** track server activity count for signaling idle flushing */
ulint last_activity_count; public: /** signalled to wake up the page_cleaner; protected by flush_list_mutex */
pthread_cond_t do_flush_list; /** broadcast when !n_flush(); protected by flush_list_mutex */
pthread_cond_t done_flush_LRU; /** broadcast when a batch completes; protected by flush_list_mutex */
pthread_cond_t done_flush_list;
/** @return number of pending LRU flush */ unsigned n_flush() const noexcept
{
mysql_mutex_assert_owner(&flush_list_mutex); return page_cleaner_status / LRU_FLUSH;
}
/** Increment the number of pending LRU flush */ inlinevoid n_flush_inc() noexcept;
/** Decrement the number of pending LRU flush */ inlinevoid n_flush_dec() noexcept;
/** @return whether flush_list flushing is active */ bool flush_list_active() const noexcept
{
mysql_mutex_assert_owner(&flush_list_mutex); return page_cleaner_status & FLUSH_LIST_ACTIVE;
}
/** @return whether the page cleaner must sleep due to being idle. Lock-freereadofpage_cleaner_idle_flag;safetocallwithout
flush_list_mutex (used by the early-return in buf_flush_ahead()). */ bool page_cleaner_idle() const noexcept
{ return page_cleaner_idle_flag;
}
/** @return whether the page cleaner may be initiating writes */ bool page_cleaner_active() const noexcept
{
mysql_mutex_assert_owner(&flush_list_mutex); return page_cleaner_status != 0;
}
/** Wake up the page cleaner if needed.
@param for_LRU whether to wake up for LRU eviction */ void page_cleaner_wakeup(bool for_LRU= false) noexcept;
/** Register whether an explicit wakeup of the page cleaner is needed */ void page_cleaner_set_idle(bool deep_sleep) noexcept
{
mysql_mutex_assert_owner(&flush_list_mutex);
page_cleaner_idle_flag= deep_sleep;
}
/** Update server last activity count */ void update_last_activity_count(ulint activity_count) noexcept
{
mysql_mutex_assert_owner(&flush_list_mutex);
last_activity_count= activity_count;
}
unsigned freed_page_clock;/*!< a sequence number used tocountthenumberofbuffer blocksremovedfromtheendof theLRUlist;NOTEthatthis countermaywraparoundat4 billion!Athreadisallowed toreadthisforheuristic purposeswithoutholdingany
mutex or latch */ /** Cleared when buf_LRU_get_free_block() fails. Setwheneverthefreelistgrows,alongwithabroadcastofdone_free.
Protected by buf_pool.mutex. */
Atomic_relaxed<bool> try_LRU_scan;
private: /** Whether we have warned to be running out of buffer pool; onlymodifiedbybuf_flush_page_cleaner():
set while holding mutex, cleared while holding flush_list_mutex */
Atomic_relaxed<bool> LRU_warned;
#ifdefined __linux__ || defined __FreeBSD__ public: /** The value of innodb_buffer_pool_in_core_dump */
my_bool in_core_dump; private: #endif
/** withdrawn blocks during resize() */
UT_LIST_BASE_NODE_T(buf_page_t) withdrawn;
public: /** list of blocks available for allocate() */
UT_LIST_BASE_NODE_T(buf_page_t) free;
/** broadcast each time when the free list grows or try_LRU_scan is set;
protected by mutex */
pthread_cond_t done_free;
/** "hazard pointer" used during scan of LRU while doing
LRU list batch. Protected by buf_pool_t::mutex. */
LRUHp lru_hp;
/** Iterator used to scan the LRU list when searching for
replacable victim. Protected by buf_pool_t::mutex. */
LRUItr lru_scan_itr;
UT_LIST_BASE_NODE_T(buf_page_t) LRU; /*!< base node of the LRU list */
buf_page_t* LRU_old; /*!< pointer to the about LRU_old_ratio/BUF_LRU_OLD_RATIO_DIV oldestblocksintheLRUlist; NULLifLRUlengthlessthan BUF_LRU_OLD_MIN_LEN; NOTE:whenLRU_old!=NULL,itslength
should always equal LRU_old_len */
ulint LRU_old_len; /*!< length of the LRU list from theblocktowhichLRU_oldpoints onward,includingthatblock; seebuf0lru.ccfortherestrictions onthisvalue;0ifLRU_old==NULL; NOTE:LRU_old_lenmustbeadjusted
whenever LRU_old shrinks or grows! */
UT_LIST_BASE_NODE_T(buf_block_t) unzip_LRU; /*!< base node of the
unzip_LRU list */
/** Try to allocate a block. @returnabufferblock
@retval nullptr if no blocks are available */
buf_block_t *allocate() noexcept; /** Remove a block from flush_list.
@param bpage buffer pool page */ void delete_from_flush_list(buf_page_t *bpage) noexcept;
/** Prepare to insert a modified blcok into flush_list. @paramlsnstartLSNofthemini-transaction
@return insert position for insert_into_flush_list() */ inline buf_page_t *prepare_insert_into_flush_list(lsn_t lsn) noexcept;
/** Insert a modified block into the flush list. @paramprevinsertposition(fromprepare_insert_into_flush_list()) @paramblockmodifiedblock
@param lsn start LSN of the mini-transaction that modified the block */ inlinevoid insert_into_flush_list(buf_page_t *prev, buf_block_t *block,
lsn_t lsn) noexcept;
/** Free a page whose underlying file page has been freed. */
ATTRIBUTE_COLD void release_freed_page(buf_page_t *bpage) noexcept;
/** Issue a warning that we could not free up buffer pool pages. */
ATTRIBUTE_COLD void LRU_warn() noexcept;
/** Print buffer pool flush state information. */
ATTRIBUTE_COLD void print_flush_info() const noexcept;
/** Collect buffer pool metadata.
@param pool_info buffer pool metadata */ void get_info(buf_pool_info_t *pool_info) noexcept;
private: /** Temporary memory for page_compressed and encrypted I/O */ struct io_buf_t
{ /** number of elements in slots[] */
ulint n_slots; /** array of slots */
buf_tmp_buffer_t *slots;
/** Set oldest_modification when adding to buf_pool.flush_list */ inlinevoid buf_page_t::set_oldest_modification(lsn_t lsn) noexcept
{
mysql_mutex_assert_owner(&buf_pool.flush_list_mutex);
ut_ad(oldest_modification() <= 1);
ut_ad(lsn > 2);
oldest_modification_= lsn;
}
/** Clear oldest_modification after removing from buf_pool.flush_list */ inlinevoid buf_page_t::clear_oldest_modification() noexcept
{ #ifdef SAFE_MUTEX if (oldest_modification() != 2)
mysql_mutex_assert_owner(&buf_pool.flush_list_mutex); #endif/* SAFE_MUTEX */
ut_d(constauto s= state());
ut_ad(s >= REMOVE_HASH);
ut_ad(oldest_modification());
ut_ad(!list.prev);
ut_ad(!list.next); /* We must use release memory order to guarantee that callers of oldest_modification_acquire()willobservetheblockas
being detached from buf_pool.flush_list, after reading the value 0. */
oldest_modification_.store(0, std::memory_order_release);
}
/** @return whether the block can be relocated in memory.
The block can be dirty, but it must not be I/O-fixed or bufferfixed. */ inlinebool buf_page_t::can_relocate() const noexcept
{
mysql_mutex_assert_owner(&buf_pool.mutex); constauto f= state();
ut_ad(f >= FREED);
ut_ad(in_LRU_list); return (f == FREED || (f < READ_FIX && !(f & ~LRU_MASK))) &&
!lock.is_locked_or_waiting();
}
/** @return whether the block has been flagged old in buf_pool.LRU */ inlinebool buf_page_t::is_old() const noexcept
{
mysql_mutex_assert_owner(&buf_pool.mutex);
ut_ad(in_file());
ut_ad(in_LRU_list); return old;
}
/** Set whether a block is old in buf_pool.LRU */ inlinevoid buf_page_t::set_old(bool old) noexcept
{
mysql_mutex_assert_owner(&buf_pool.mutex);
ut_ad(in_LRU_list);
#ifdef UNIV_LRU_DEBUG
ut_a((buf_pool.LRU_old_len == 0) == (buf_pool.LRU_old == nullptr)); /* If a block is flagged "old", the LRU_old list must exist. */
ut_a(!old || buf_pool.LRU_old);
/** Select from where to start a scan. If we have scanned toodeepintotheLRUlistitresetsthevaluetothetail oftheLRUlist.
@return buf_page_t from where to start scan. */ inline buf_page_t *LRUItr::start() noexcept
{
mysql_mutex_assert_owner(m_mutex);
if (!m_hp || m_hp->old)
m_hp= UT_LIST_GET_LAST(buf_pool.LRU);
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.36Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.