/** Create or restore the doublewrite buffer in the TRX_SYS page.
@return whether the operation succeeded */ bool buf_dblwr_t::create() noexcept
{ if (is_created()) returntrue;
if (mach_read_from_4(fseg_header + FSEG_HEADER_SIZE) ==
TRX_SYS_DOUBLEWRITE_MAGIC_N)
{ /* The doublewrite buffer has already been created: just read in
some numbers */
init(TRX_SYS_DOUBLEWRITE + trx_sys_block->page.frame);
mtr.commit(); returntrue;
}
if (UT_LIST_GET_FIRST(fil_system.sys_space->chain)->size < 3 * size)
{
sql_print_error("InnoDB: Cannot create doublewrite buffer: " "the first file in innodb_data_file_path must be at least " "%zuM.", 3 * (size >> (20U - srv_page_size_shift)));
fail:
mtr.commit(); returnfalse;
} else
{
buf_block_t *b= fseg_create(fil_system.sys_space,
TRX_SYS_DOUBLEWRITE + TRX_SYS_DOUBLEWRITE_FSEG,
&mtr, &err, false, trx_sys_block); if (!b)
{
sql_print_error("InnoDB: Cannot create doublewrite buffer: %s",
ut_strerr(err)); goto fail;
}
sql_print_information("InnoDB: Doublewrite buffer not found:" " creating new");
/* FIXME: The doublewrite buffer should not exist in theInnoDBsystemtablespacefileinthefirstplace. Itcouldbelocatedinseparateoptionalfile(s)ina
user-specified location. */
}
mtr_t init_mtr{nullptr};
init_mtr.start();
for (uint32_t prev_page_no= 0, i= 0, extent_size= FSP_EXTENT_SIZE;
i < 2 * size + extent_size / 2; i++)
{
buf_block_t *new_block=
fseg_alloc_free_page_general(fseg_header, prev_page_no + 1, FSP_UP, false, &mtr, &init_mtr, &err); if (!new_block)
{
sql_print_error("InnoDB: Cannot create doublewrite buffer: " " you must increase your tablespace size." " Cannot continue operation."); /* This may essentially corrupt the doublewrite buffer.However,usuallythedoublewritebuffer iscreatedatdatabaseinitialization,andit shouldnotmatter(justremoveallnewlycreated
InnoDB files and restart). */
mtr.commit(); returnfalse;
}
const page_id_t id= new_block->page.id(); /* Normally, allocated pages will be modified further. However, thepagesofthedoublewritebufferarejustdummystorage,not
covered by the write-ahead log. */
ut_ad(init_mtr.get_savepoint() == 1);
ut_ad(init_mtr.m_memo[0].object == new_block);
ut_ad(init_mtr.m_memo[0].type == MTR_MEMO_PAGE_X_MODIFY);
new_block->page.fix();
init_mtr.m_memo[0].type= MTR_MEMO_PAGE_X_FIX;
init_mtr.rollback_to_savepoint(0, 1);
init_mtr.m_log.erase();
mysql_mutex_lock(&buf_pool.mutex);
new_block->page.unfix();
ut_d(bool freed=) buf_LRU_free_page(&new_block->page, true);
ut_ad(freed);
mysql_mutex_unlock(&buf_pool.mutex);
if (i == size / 2)
ut_a(id.page_no() == size); elseif (i == size / 2 + size)
ut_a(id.page_no() == 2 * size); elseif (i > size / 2)
ut_a(id.page_no() == prev_page_no + 1);
prev_page_no= id.page_no();
}
/* We do the file i/o past the buffer pool */
byte *read_buf= static_cast<byte*>(aligned_malloc(srv_page_size,
srv_page_size)); /* Read the TRX_SYS header to check if we are using the doublewrite buffer */
dberr_t err= os_file_read(IORequestRead, file, read_buf,
TRX_SYS_PAGE_NO << srv_page_size_shift,
srv_page_size, nullptr);
if (err != DB_SUCCESS)
{
sql_print_error("InnoDB: Failed to read the system tablespace" " header page");
func_exit:
aligned_free(read_buf); return err;
}
/* TRX_SYS_PAGE_NO is not encrypted see fil_crypt_rotate_page() */ if (mach_read_from_4(TRX_SYS_DOUBLEWRITE_MAGIC + TRX_SYS_DOUBLEWRITE +
read_buf) != TRX_SYS_DOUBLEWRITE_MAGIC_N)
{ /* There is no doublewrite buffer initialized in the TRX_SYS page. Thisshouldnormallynotbepossible;thedoublewritebuffershould
be initialized when creating the database. */
err= DB_SUCCESS; goto func_exit;
}
auto write_buf= active_slot->write_buf; /* Read the pages from the doublewrite buffer to memory */
err= os_file_read(IORequestRead, file, write_buf,
block1.page_no() << srv_page_size_shift,
size << srv_page_size_shift, nullptr);
if (err != DB_SUCCESS)
{
sql_print_error("InnoDB: Failed to read" " the first double write buffer extent"); goto func_exit;
}
err= os_file_read(IORequestRead, file,
write_buf + (size << srv_page_size_shift),
block2.page_no() << srv_page_size_shift,
size << srv_page_size_shift, nullptr); if (err != DB_SUCCESS)
{
sql_print_error("InnoDB: Failed to read" " the second double write buffer extent"); goto func_exit;
}
byte *page= write_buf;
if (UNIV_UNLIKELY(upgrade_to_innodb_file_per_table))
{
sql_print_information("InnoDB: Resetting space id's in " "the doublewrite buffer");
for (ulint i= 0; i < size * 2; i++, page += srv_page_size)
{
memset(page + FIL_PAGE_SPACE_ID, 0, 4); /* For pre-MySQL-4.1 innodb_checksum_algorithm=innodb, we do not need to calculatenewchecksumsforthepagesbecausethefield .._SPACE_IDdoesnotaffectthem.Writethepagebacktowhere
we read it from. */ const ulint source_page_no= i < size
? block1.page_no() + i
: block2.page_no() + i - size;
err= os_file_write(IORequestWrite, path, file, page,
source_page_no << srv_page_size_shift, srv_page_size); if (err != DB_SUCCESS)
{
sql_print_error("InnoDB: Failed to upgrade the double write buffer"); goto func_exit;
}
}
os_file_flush(file);
} else
{
alignas(8) char checkpoint[8];
mach_write_to_8(checkpoint, log_sys.last_checkpoint_lsn); for (auto i= size * 2; i--; page += srv_page_size) if (memcmp_aligned<8>(page + FIL_PAGE_LSN, checkpoint, 8) >= 0) /* Valid pages are not older than the log checkpoint. */
recv_sys.dblwr.add(page);
}
err= DB_SUCCESS; goto func_exit;
}
/** Process and remove the double write buffer pages for all tablespaces. */ void buf_dblwr_t::recover() noexcept
{
ut_ad(log_sys.last_checkpoint_lsn); if (!is_created()) return; const lsn_t max_lsn{log_sys.get_flushed_lsn(std::memory_order_relaxed)};
ut_ad(recv_sys.scanned_lsn == max_lsn ||
(recv_sys.rpo && recv_sys.rpo < max_lsn));
ut_ad(recv_sys.scanned_lsn >= recv_sys.lsn);
std::deque<byte*> deferred_pages; for (recv_dblwr_t::list::iterator i= recv_sys.dblwr.pages.begin();
i != recv_sys.dblwr.pages.end(); ++i, ++page_no_dblwr)
{ const page_t *const page= *i; const uint32_t page_no= page_get_page_no(page); const lsn_t lsn= mach_read_from_8(page + FIL_PAGE_LSN); if (log_sys.last_checkpoint_lsn > lsn || lsn > max_lsn) /* Pages written before or after the recovery range are not usable. */ continue; const uint32_t space_id= page_get_space_id(page); const page_id_t page_id(space_id, page_no);
fil_space_t *space= fil_space_t::get(space_id);
if (!space)
{ /* These pages does not appear to belong to any tablespace. Thereisapossibilitythatthispagecouldbe encrypted/compressedusingfull_crc32format. Ifinnodbencountersanycorruptedencrypted/compressed pageduringrecoverythenInnoDBshouldusethispageto findthevalidpage.
See find_encrypted_page()/find_page_compressed() */
deferred_pages.push_back(*i); continue;
}
if (UNIV_UNLIKELY(page_no >= space->get_size()))
{ /* Do not report the warning for undo tablespaces, because they
can be truncated in place. */ if (!srv_is_undo_tablespace(space_id))
sql_print_warning("InnoDB: A copy of page " "[page id: space=" UINT32PF ", page number=" UINT32PF "]" " in the doublewrite buffer slot " UINT32PF " is beyond the end of %s (" UINT32PF " pages)",
page_id.space(), page_id.page_no(),
page_no_dblwr, space->chain.start->name,
space->size);
next_page:
space->release(); continue;
}
/* We want to ensure that for partial reads the unread portion of
the page is NUL. */
memset(read_buf, 0x0, physical_size);
/* Read in the actual page from the file */
fil_io_t fio= space->io(IORequest(IORequest::DBLWR_RECOVER),
os_offset_t{page_no} * physical_size,
physical_size, read_buf);
if (UNIV_UNLIKELY(fio.err != DB_SUCCESS))
sql_print_warning("InnoDB: Double write buffer recovery: " "[page id: space=" UINT32PF ", page number=" UINT32PF "]" " ('%s') read failed with error: %s",
page_id.space(), page_id.page_no(), fio.node->name,
ut_strerr(fio.err)); elseif (buf_is_zeroes(span<const byte>(read_buf, physical_size)))
{ /* We will check if the copy in the doublewrite buffer is valid.Ifnot,wewillignorethispage(thereshouldberedo
log records to initialize it). */
} elseif (recv_sys.dblwr.validate_page(page_id, max_lsn, space,
read_buf, buf)) goto next_page; else /* We intentionally skip this message for all-zero pages. */
sql_print_information("InnoDB: Trying to recover page " "[page id: space=" UINT32PF ", page number=" UINT32PF "]" " from the doublewrite buffer.",
page_id.space(), page_id.page_no());
if (const byte *page=
recv_sys.dblwr.find_page(page_id, max_lsn, space, buf))
{ /* Write the good page from the doublewrite buffer to the intended
position. */
space->reacquire();
fio= space->io(IORequestWrite,
os_offset_t{page_id.page_no()} * physical_size,
physical_size, const_cast<byte*>(page));
if (fio.err == DB_SUCCESS)
sql_print_information("InnoDB: Recovered page " "[page id: space=" UINT32PF ", page number=" UINT32PF "]" " to '%s' from the doublewrite buffer.",
page_id.space(), page_id.page_no(),
fio.node->name);
}
goto next_page;
}
recv_sys.dblwr.pages.clear(); for (byte *page : deferred_pages)
recv_sys.dblwr.pages.push_back(page);
fil_flush_file_spaces();
aligned_free(read_buf);
}
/** Free the doublewrite buffer. */ void buf_dblwr_t::close() noexcept
{ if (!active_slot) return;
if (!--flush_slot->reserved)
{
mysql_mutex_unlock(&mutex); /* This will finish the batch. Sync data files to the disk. */
fil_flush_file_spaces();
mysql_mutex_lock(&mutex);
/* We can now reuse the doublewrite memory buffer: */
flush_slot->first_free= 0;
batch_running= false;
pthread_cond_broadcast(&cond);
}
/** Check the LSN values on the page with which this block is associated. */ staticvoid buf_dblwr_check_block(const buf_page_t *bpage) noexcept
{
ut_ad(bpage->in_file()); const page_t *page= bpage->frame;
ut_ad(page);
switch (fil_page_get_type(page)) { case FIL_PAGE_INDEX: case FIL_PAGE_TYPE_INSTANT: case FIL_PAGE_RTREE: if (page_is_comp(page))
{ if (page_simple_validate_new(page)) return;
} elseif (page_simple_validate_old(page)) return; /* While it is possible that this is not an index page but just happenstohavewronglysetFIL_PAGE_TYPE,suchpagesshouldnever bemodifiedtowithoutalsoadjustingthepagetypeduringpage allocationorbuf_flush_init_for_writing()or
fil_block_reset_type(). */
buf_page_print(page);
ib::fatal() << "Apparent corruption of an index page " << bpage->id()
<< " to be written to data file. We intentionally crash" " the server to prevent corrupt data from ending up in" " data files.";
}
} #endif/* UNIV_DEBUG */
/* Make the doublewrite durable. Note: The doublewrite buffer is alwaysinthefirstfileofthesystemtablespace.Wewillnot botheraboutfil_system.unflushed_spaces,whichcanresultina redundantcallduringfil_flush_file_spaces()in log_checkpoint().Writestothesystemtablespaceshouldberare, exceptwhenexecutingDDLorusingthenon-defaultsettings
innodb_file_per_table=OFF or innodb_undo_tablespaces=0. */
os_file_flush(request.node->handle);
/* The writes have been flushed to disk now and in recovery we will
find them in the doublewrite buffer blocks. Next, write the data pages. */ for (ulint i= 0, first_free= flush_slot->first_free; i < first_free; i++)
{ auto e= flush_slot->buf_block_arr[i];
buf_page_t* bpage= e.request.bpage;
ut_ad(bpage->in_file());
/** Flush possible buffered writes to persistent storage. Itisveryimportanttocallthisfunctionafterabatchofwriteshasbeen posted,andalsowhenwemayhavetowaitforapagelatch!
Otherwise a deadlock of threads can occur. */ void buf_dblwr_t::flush_buffered_writes() noexcept
{
mysql_mutex_lock(&mutex);
/* "frame" is at least 1024-byte aligned for ROW_FORMAT=COMPRESSED pages,
and at least srv_page_size (4096-byte) for everything else. */
memcpy_aligned<UNIV_ZIP_SIZE_MIN>(p, get_frame(request), size); /* fil_page_compress() for page_compressed guarantees 256-byte alignment */
memset_aligned<256>(p + size, 0, srv_page_size - size); /* FIXME: Inform the compiler that "size" and "srv_page_size - size" areintegermultiplesof256,sotheabovecantranslateintosimple SIMDinstructions.Currently,wemakenosuchassumptionsaboutthe
non-pointer parameters that are passed to the _aligned templates. */
ut_ad(!request.bpage->zip_size() || request.bpage->zip_size() == size);
ut_ad(active_slot->reserved == active_slot->first_free);
ut_ad(active_slot->reserved < buf_size); new (active_slot->buf_block_arr + active_slot->first_free++)
element{request.doublewritten(), size};
active_slot->reserved= active_slot->first_free;
if (active_slot->first_free != buf_size ||
!flush_buffered_writes(buf_size / 2))
mysql_mutex_unlock(&mutex);
}
Messung V0.5 in Prozent
¤ Dauer der Verarbeitung: 0.13 Sekunden
(vorverarbeitet am 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.