/** The recovery system */
recv_sys_t recv_sys; /** 0 or the first LSN that would conflict with innodb_log_recovery_target */ static lsn_t recv_sys_rpo_exceeded; /** TRUE when recv_init_crash_recovery() has been called. */ bool recv_needed_recovery; #ifdef UNIV_DEBUG /** TRUE if writing to the redo log (mtr_commit) is forbidden.
Protected by log_sys.latch. */ bool recv_no_log_write = false; #endif/* UNIV_DEBUG */
/** Error from mlog_decode_len() */
constexpr uint32_t MLOG_DECODE_ERROR= ~0U;
/** Decode the length of a variable-length encoded integer. @paramfirstfirstbyteoftheencodedinteger
@return the length, in bytes */ static uint8_t mlog_decode_varint_length(uint8_t first) noexcept
{ #if __cpp_lib_bitops >= 201907L /* C++20 #include <bit> */ return1 + std::countl_one(first); #elifdefined __GNUC__ && !defined __s390__ && !defined __m68k__ && !defined __riscv && !defined __sparc__ /* Make use of the MIPS/POWER/ARM clz instruction or equivalent, suchasthex86BitScanReverse(bsr)orlzcnt(repbsr). Ons390x,68k,RISC-VorSPARC,thiswouldtranslateintoacall
to a library function. We prefer a simple loop in that case. */ return first == 0xff ? first : uint8_t(__builtin_clz(uint8_t(~first)) - 23); #else int len{1}; for (int f{first}; f & 0x80; len++, f*= 2); return uint8_t(len); #endif
}
/** Decode a variable-encoded integer @paramloglogrecordsnippet @returnthedecodedinteger
@retval MLOG_DECODE_ERROR on error */
ATTRIBUTE_NOINLINE static uint32_t mlog_decode_varint(const byte *log) noexcept
{
uint32_t i{*log}; if (i < MIN_2BYTE) return i; if (i < 0xc0) return MIN_2BYTE + ((i & ~0xc0) << 8 | log[1]); if (i < 0xe0) return MIN_3BYTE + ((i & ~0xe0) << 16 | uint32_t{log[1]} << 8 | log[2]); if (i < 0xf0) return MIN_4BYTE + ((i & ~0xf0) << 24 | uint32_t{log[1]} << 16 |
uint32_t{log[2]} << 8 | log[3]); if (i == 0xf0)
{
i= mach_read_from_4(log + 1); if (i <= ~MIN_5BYTE) return MIN_5BYTE + i;
} return MLOG_DECODE_ERROR;
}
/** Decode the length of a record. @parambuflogrecord(willbeadvancedafterthelength) @paramsizetotalsizeoftherecord
@return the log record payload after the encoded length */
ATTRIBUTE_NOINLINE const byte *mtr_t::parse_length(const byte *l, uint32_t *size) noexcept
{
uint32_t s(*l++ & 0xf); if (!s)
{
s= mlog_decode_varint_length(*l); const uint32_t addlen{mlog_decode_varint(l)};
ut_ad(addlen != MLOG_DECODE_ERROR);
l+= s;
s= addlen + 15 - s;
}
*size= s; return l;
}
/** Flag the log corrupted during recovery. */
ATTRIBUTE_COLD staticvoid log_set_corrupt() noexcept
{
mysql_mutex_lock(&recv_sys.mutex);
recv_sys.set_corrupt_log();
mysql_mutex_unlock(&recv_sys.mutex);
}
/** Stored physical log record */ struct log_phys_t : public log_rec_t
{ /** start LSN of the mini-transaction (not necessarily of this record) */ const lsn_t start_lsn; private: /** @return the start of length and data */ const byte *start() const
{ return my_assume_aligned<sizeof(size_t)>
(reinterpret_cast<const byte*>(&start_lsn + 1));
} /** @return the start of length and data */
byte *start()
{ returnconst_cast<byte*>(const_cast<const log_phys_t*>(this)->start()); } /** @return the length of the following record */
uint16_t len() const { uint16_t i; memcpy(&i, start(), 2); return i; }
/** @return start of the log records */
byte *begin() { return start() + 2; } /** @return end of the log records */
byte *end() { byte *e= begin() + len(); ut_ad(!*e); return e; } public: /** @return start of the log records */ const byte *begin() const { returnconst_cast<log_phys_t*>(this)->begin(); } /** @return end of the log records */ const byte *end() const { returnconst_cast<log_phys_t*>(this)->end(); }
/** Determine the allocated size of the object. @paramlenlengthofrecs,excludingterminatingNULbyte
@return the total allocation size */ staticinline size_t alloc_size(size_t len);
/** The status of apply() */ enum apply_status { /** The page was not affected */
APPLIED_NO= 0, /** The page was modified */
APPLIED_YES, /** The page was modified, affecting the encryption parameters */
APPLIED_TO_ENCRYPTION, /** The page was modified, affecting the tablespace header */
APPLIED_TO_FSP_HEADER, /** The page was found to be corrupted */
APPLIED_CORRUPTED,
};
/** Apply log to a page frame. @param[in,out]blockbufferblock @param[in,out]last_offsetlastbyteoffset,forsame_pagerecords
@return whether any log was applied to the page */
apply_status apply(const buf_block_t &block, uint16_t &last_offset) const
{ const byte * const recs= begin();
byte *const frame= block.page.zip.data
? block.page.zip.data : block.page.frame; const size_t size= block.physical_size();
apply_status applied= APPLIED_NO;
/** Tablespace item during recovery */ struct file_name_t { /** Tablespace file name (FILE_MODIFY) */
std::string name; /** Tablespace object (NULL if not valid or not found) */
fil_space_t* space = nullptr;
/** Status of the tablespace */
fil_status status;
/** Log sequence number of a FILE_CREATE record, or 0 */
lsn_t create_lsn = 0;
/** FSP_SIZE of tablespace */
uint32_t size = 0;
/** Freed pages of tablespace */
range_set freed_ranges;
/** Dummy flags before they have been read from the .ibd file */ static constexpr uint32_t initial_flags = FSP_FLAGS_FCRC32_MASK_MARKER; /** FSP_SPACE_FLAGS of tablespace */
uint32_t flags = initial_flags;
/** Remove the freed pages */ void remove_freed_page(uint32_t page_no)
{ if (freed_ranges.empty()) return;
freed_ranges.remove_value(page_no);
}
};
/** Map of dirty tablespaces during recovery */ typedef std::map<
uint32_t,
file_name_t,
std::less<uint32_t>,
ut_allocator<std::pair<const uint32_t, file_name_t> > > recv_spaces_t;
static recv_spaces_t recv_spaces;
/** The last parsed FILE_RENAME records */ static std::map<uint32_t,std::string> renamed_spaces;
/** Files for which fil_ibd_load() returned FIL_LOAD_DEFER */ staticstruct
{ /** Maintains the last opened defer file name along with lsn */ struct item
{ /** Log sequence number of latest add() called by fil_name_process() */
lsn_t lsn; /** File name from the FILE_ record */
std::string file_name; /** whether a FILE_DELETE record was encountered */ mutablebool deleted;
};
/** Add the deferred space only if it is latest one @paramspacespaceidentifier @paramf_namefilename
@param lsn log sequence number of the FILE_ record */ void add(uint32_t space, const std::string &f_name, lsn_t lsn)
{
mysql_mutex_assert_owner(&recv_sys.mutex); constchar *filename= f_name.c_str();
if (srv_operation == SRV_OPERATION_RESTORE)
{ /* Replace absolute DATA DIRECTORY file paths with
short names relative to the backup directory. */ if (constchar *name= strrchr(filename, '/'))
{ while (--name > filename && *name != '/'); if (name > filename)
filename= name + 1;
}
}
/* The file name must be unique. Keep the one with the latest LSN. */ auto d= defers.begin();
while (d != defers.end())
{ if (d->second.file_name != defer.file_name)
++d; elseif (d->first == space)
{ /* Neither the file name nor the tablespace ID changed.
Update the LSN if needed. */ if (d->second.lsn < lsn)
d->second.lsn= lsn; return;
} elseif (d->second.lsn < lsn)
{ /* Reset the old tablespace name in recovered spaces list */
recv_spaces_t::iterator it{recv_spaces.find(d->first)}; if (it != recv_spaces.end() &&
it->second.name == d->second.file_name)
it->second.name = "";
defers.erase(d++);
} else
{
ut_ad(d->second.lsn != lsn); return; /* A later tablespace already has this name. */
}
}
auto p= defers.emplace(space, defer); if (!p.second && p.first->second.lsn <= lsn)
{
p.first->second.lsn= lsn;
p.first->second.file_name= defer.file_name;
} /* Add the newly added deferred space and change the file name */
recv_spaces_t::iterator it{recv_spaces.find(space)}; if (it != recv_spaces.end())
it->second.name = defer.file_name;
}
/** Look up a tablespace that was found corrupted during recovery. @paramidtablespaceid @returntablespacewhosecreationwasdeferred
@retval nullptr if no such tablespace was found */
item *find(uint32_t id)
{
mysql_mutex_assert_owner(&recv_sys.mutex); auto it= defers.find(id); if (it != defers.end()) return &it->second; return nullptr;
}
/** Report an operation to create, delete, or rename a file during backup. @param[in]space_idtablespaceidentifier @param[in]typeredologtype @param[in]namefilename(notNUL-terminated) @param[in]lenlengthofname,inbytes @param[in]new_namenewfilename(NULLifnotrename)
@param[in] new_len length of new_name, in bytes (0 if NULL) */ void (*log_file_op)(uint32_t space_id, int type, const byte* name, size_t len, const byte* new_name, size_t new_len);
void (*undo_space_trunc)(uint32_t space_id);
void (*first_page_init)(uint32_t space_id);
/** Information about initializing page contents during redo log processing.
FIXME: Rely on recv_sys.pages! */ class mlog_init_t
{ using map= std::map<const page_id_t, lsn_t,
std::less<const page_id_t>,
ut_allocator<std::pair<const page_id_t, lsn_t>>>; /** Map of page initialization operations.
FIXME: Merge this to recv_sys.pages! */
map inits;
/** Iterator to the last add() or will_avoid_read(), for speeding up
will_avoid_read(). */
map::iterator i; public: /** Constructor */
mlog_init_t() : i(inits.end()) {}
/** Record that a page will be initialized by the redo log. @parampage_idpageidentifier @paramlsnlogsequencenumber
@return whether the state was changed */ bool add(const page_id_t page_id, lsn_t lsn)
{
mysql_mutex_assert_owner(&recv_sys.mutex);
std::pair<map::iterator, bool> p=
inits.emplace(map::value_type{page_id, lsn}); if (p.second) returntrue; if (p.first->second >= lsn) returnfalse;
p.first->second= lsn;
i= p.first; returntrue;
}
/** Get the last initialization lsn of a page. @parampage_idpageidentifier @returnthelatestpageinitialization;
not valid after releasing recv_sys.mutex. */
lsn_t last(page_id_t page_id)
{
mysql_mutex_assert_owner(&recv_sys.mutex); return inits.find(page_id)->second;
}
/** Determine if a page will be initialized or freed after a time. @parampage_idpageidentifier @paramlsnlogsequencenumber
@return whether page_id will be freed or initialized after lsn */ bool will_avoid_read(page_id_t page_id, lsn_t lsn)
{
mysql_mutex_assert_owner(&recv_sys.mutex); if (i != inits.end() && i->first == page_id) return i->second > lsn;
i= inits.lower_bound(page_id); return i != inits.end() && i->first == page_id && i->second > lsn;
}
/** Clear the data structure */ void clear() { inits.clear(); i= inits.end(); }
};
static mlog_init_t mlog_init;
/** Try to recover a tablespace that was not readable earlier @parampiteratortothepage @paramnametablespacefilename @paramfree_blocksparebufferblock @returnrecoveredtablespace
@retval nullptr if recovery failed */
fil_space_t *recv_sys_t::recover_deferred(const recv_sys_t::map::iterator &p, const std::string &name,
buf_block_t *&free_block)
{
mysql_mutex_assert_owner(&mutex);
if (page_id_t{space_id, page_no} == p->first && size >= 4 &&
fil_space_t::is_valid_flags(flags, space_id) &&
fil_space_t::logical_size(flags) == srv_page_size)
{
fil_space_t *space= deferred_spaces.create(it, name, flags,
fil_space_read_crypt_data
(fil_space_t::zip_size(flags),
page), size); if (!space) goto release_and_fail;
space->free_limit= fsp_header_get_field(page, FSP_FREE_LIMIT);
space->free_len= flst_get_len(FSP_HEADER_OFFSET + FSP_FREE + page);
fil_node_t *node= UT_LIST_GET_FIRST(space->chain);
node->deferred= true;
mysql_mutex_unlock(&fil_system.mutex); if (!space->acquire()) goto release_and_fail;
fil_names_dirty(space); constbool is_compressed= fil_space_t::is_compressed(flags); #ifdef _WIN32 constbool is_sparse= is_compressed; if (is_compressed)
os_file_set_sparse_win32(node->handle); #else constbool is_sparse= is_compressed &&
DB_SUCCESS == os_file_punch_hole(node->handle, 0, 4096) &&
!my_test_if_thinly_provisioned(node->handle); #endif /* Mimic fil_node_t::read_page0() in case the file exists and
has already been extended to a larger size. */
ut_ad(node->size == size); const os_offset_t file_size= os_file_get_size(node->handle); if (file_size != os_offset_t(-1))
{ const uint32_t n_pages=
uint32_t(file_size / fil_space_t::physical_size(flags)); if (n_pages > size)
{
mysql_mutex_lock(&fil_system.mutex);
space->size= node->size= n_pages;
space->set_committed_size();
mysql_mutex_unlock(&fil_system.mutex); goto size_set;
}
} if (!os_file_set_size(node->name, node->handle,
(size * fil_space_t::physical_size(flags)) &
~4095ULL, is_sparse))
{
space->release(); goto release_and_fail;
}
size_set:
node->deferred= false;
it->second.space= space;
block->page.lock.x_unlock();
p->second.being_processed= -1; return space;
}
release_and_fail:
block->page.lock.x_unlock();
}
fail:
ib::error() << "Cannot apply log to " << p->first
<< " of corrupted file '" << name << "'"; return nullptr;
}
/** Process a record that indicates that a tablespace is beingshrunkinsize. @parampage_idfirstpageidentifierthatisnotinthefile
@param lsn log sequence number of the shrink operation */
ATTRIBUTE_COLD void recv_sys_t::trim(const page_id_t page_id, lsn_t lsn)
{
DBUG_ENTER("recv_sys_t::trim");
DBUG_LOG("ib_log", "discarding log beyond end of tablespace "
<< page_id << " before LSN " << lsn);
mysql_mutex_assert_owner(&mutex); if (pages_it != pages.end() && pages_it->first.space() == page_id.space())
pages_it= pages.end(); for (recv_sys_t::map::iterator p = pages.lower_bound(page_id);
p != pages.end() && p->first.space() == page_id.space();)
{
recv_sys_t::map::iterator r = p++; if (r->second.trim(lsn))
{
ut_ad(!r->second.being_processed);
pages.erase(r);
}
}
DBUG_VOID_RETURN;
}
/** Apply a FILE_DELETE record to a tablespace that has not been loaded yet. */
ATTRIBUTE_COLD staticvoid fil_delete_apply(uint32_t id, constchar *name)
noexcept
{ if (log_sys.archive)
{
fil_space_t *space{nullptr}; if (fil_ibd_load(id, name, space) == FIL_LOAD_OK)
{
ut_ad(space);
fil_delete_apply(space); return;
}
ut_ad(!space);
}
}
/** Process a file name from a FILE_* record. @param[in]namefilename @param[in]lenlengthofthefilename @param[in]space_idthetablespaceID @param[in]ftypeFILE_CREATE,FILE_MODIFY,FILE_DELETE, orFILE_RENAME @param[in]lsnlsnoftheredolog
@param[in] if_exists whether to check if the tablespace exists */ staticvoid fil_name_process(constchar *name, ulint len, uint32_t space_id,
mfile_type_t ftype, lsn_t lsn, bool if_exists)
{
ut_ad(srv_operation <= SRV_OPERATION_EXPORT_RESTORED
|| srv_operation == SRV_OPERATION_RESTORE
|| srv_operation == SRV_OPERATION_RESTORE_EXPORT);
/* We will also insert space=NULL into the map, so that furthercheckscanensurethataFILE_MODIFYrecordwas
scanned before applying any page records for the space_id. */
file_name_t& f = p.first->second;
ut_ad(!f.space || f.space->id == space_id); auto d = deferred_spaces.find(space_id); /* deferred_spaces must be a proper subset of recv_spaces. Bothhereandinrecv_validate_tablespace(),deferred_spaces.add()
may only be invoked for entries that exist in recv_spaces. */
ut_ad(!d || !p.second);
if (deleted) { /* Got FILE_DELETE */ if (d) {
d->deleted = true;
} if (p.second) {
fil_delete_apply(space_id, f.name.c_str());
} elseif (f.status != file_name_t::DELETED) {
f.status = file_name_t::DELETED; if (f.space != NULL) {
fil_delete_apply(f.space);
f.space = NULL;
}
}
ut_ad(f.space == NULL);
f.create_lsn = 0;
} elseif (d || p.second /* the first FILE_MODIFY or FILE_RENAME */
|| f.name != fname.name) { if (f.name.size() == 0) { /* Augment the recv_spaces.emplace_hint() for the FILE_MODIFYrecordthathadbeenaddedby
recv_sys_t::parse() */
f.name = fname.name;
}
fil_space_t* space;
/* Check if the tablespace file exists and contains thespace_id.Ifnot,ignorethefileafterdisplaying anote.Abortiftherearemultiplefileswiththe
same space_id. */ switch (fil_load_status s
= fil_ibd_load(space_id, fname.name.c_str(), space)) { case FIL_LOAD_OK:
ut_ad(space != NULL);
deferred_spaces.remove(space_id); if (!f.space) { if (f.size
|| f.flags != f.initial_flags) {
fil_space_set_recv_size_and_flags(
space->id, f.size, f.flags);
}
f.space = space; goto same_space;
} elseif (f.space == space) {
same_space:
f.name = fname.name;
f.status = file_name_t::NORMAL;
} else {
sql_print_error("InnoDB: Tablespace " UINT32PF " has been found" " in two places:" " '%.*s' and '%.*s'." " You must delete" " one of them.",
space_id, int(f.name.size()),
f.name.data(), int(fname.name.size()),
fname.name.data());
recv_sys.set_corrupt_fs();
} break;
case FIL_LOAD_ID_CHANGED: case FIL_LOAD_NOT_FOUND: /* No matching tablespace was found; maybe it wasrenamed,andwewillfindasubsequent
FILE_* record. */
ut_ad(space == NULL);
if (f.create_lsn) { if (d && ftype == FILE_RENAME) {
rename:
d->file_name = fname.name;
f.name = fname.name;
} break;
}
if (ftype == FILE_CREATE) {
f.create_lsn = lsn; break;
}
if (s == FIL_LOAD_ID_CHANGED) { break;
}
if (srv_force_recovery
|| srv_operation == SRV_OPERATION_RESTORE) { /* Without innodb_force_recovery, missingtablespaceswillonlybe reportedin recv_init_crash_recovery_spaces(). Enablesomemorediagnosticswhen
forcing recovery. */
sql_print_information( "InnoDB: At LSN: " LSN_PF ": unable to open file %.*s" " for tablespace " UINT32PF,
recv_sys.lsn, int(fname.name.size()),
fname.name.data(), space_id);
} break;
case FIL_LOAD_DEFER: if (d && ftype == FILE_RENAME && f.create_lsn) { goto rename;
} /* Skip the deferred spaces
when lsn is already processed */ if (!if_exists) {
deferred_spaces.add(
space_id, fname.name.c_str(), lsn);
} break; case FIL_LOAD_INVALID:
ut_ad(space == NULL); if (srv_force_recovery == 0) {
sql_print_error("InnoDB: Recovery cannot access" " file %.*s (tablespace "
UINT32PF ")", int(len), name,
space_id);
sql_print_information("InnoDB: You may set " "innodb_force_recovery=1" " to ignore this and" " possibly get a" " corrupted database.");
recv_sys.set_corrupt_fs(); break;
}
/** Calculate the checksum for a log block using the pre-10.2.2 algorithm. */ inline uint32_t log_block_calc_checksum_format_0(const byte *b)
{
uint32_t sum= 1; const byte *const end= &b[512 - 4];
for (uint32_t sh= 0; b < end; )
{
sum&= 0x7FFFFFFFUL;
sum+= uint32_t{*b} << sh++;
sum+= *b++; if (sh > 24)
sh= 0;
}
return sum;
}
/** Determine if a redo log from before MariaDB 10.2.2 is clean. @returnerrorcode @retvalDB_SUCCESSiftheredologisclean @retvalDB_CORRUPTIONiftheredologiscorrupted
@retval DB_ERROR if the redo log is not empty */
ATTRIBUTE_COLD static dberr_t recv_log_recover_pre_10_2()
{
uint64_t max_no= 0;
ut_ad(log_sys.format == 0);
/** Offset of the first checkpoint checksum */
constexpr uint CHECKSUM_1= 288; /** Offset of the second checkpoint checksum */
constexpr uint CHECKSUM_2= CHECKSUM_1 + 4; /** the checkpoint LSN field */
constexpr uint CHECKPOINT_LSN= 8; /** Most significant bits of the checkpoint offset */
constexpr uint OFFS_HI= CHECKSUM_2 + 12; /** Least significant bits of the checkpoint offset */
constexpr uint OFFS_LO= 16;
constchar *uag= srv_operation == SRV_OPERATION_NORMAL
? "InnoDB: Upgrade after a crash is not supported."
: "mariadb-backup --prepare is not possible.";
if (log_block_calc_checksum_format_0(buf) !=
mach_read_from_4(my_assume_aligned<4>(buf + 508)) &&
!log_crypt_101_read_block(buf, log_sys.last_checkpoint_lsn))
{
sql_print_error("%s%s, and it appears corrupted.", uag, pre_10_2); return DB_CORRUPTION;
}
if (mach_read_from_2(buf + 4) == (source_offset & 511)) return DB_SUCCESS;
if (buf[20 + 32 * 9] == 2)
sql_print_error("InnoDB: Cannot decrypt log for upgrading." " The encrypted log was created before MariaDB 10.2.2."); else
sql_print_error("%s%s. You must start up and shut down" " MariaDB 10.1 or MySQL 5.6 or earlier" " on the data directory.",
uag, pre_10_2);
return DB_ERROR;
}
/** Determine if a redo log from MariaDB 10.2.2, 10.3, 10.4, or 10.5 is clean. @paramlsn_offsetcheckpointLSNoffset @returnerrorcode @retvalDB_SUCCESSiftheredologisclean @retvalDB_CORRUPTIONiftheredologiscorrupted
@retval DB_ERROR if the redo log is not empty */ static dberr_t recv_log_recover_10_5(lsn_t lsn_offset)
{
byte *buf= const_cast<byte*>(field_ref_zero);
if (UNIV_UNLIKELY(log_sys.buf_size > i->second.end - i->first))
log_sys.buf_size= unsigned(i->second.end - i->first);
if (!log_sys.attach(file, i->second.end - i->first +
log_t::START_OFFSET, log_t::READ_ONLY))
{
os_file_close(file); return DB_ERROR;
} const dberr_t err=
find_checkpoint_archived(i->first, !read_only && i != start); if (!uint16_t(~last_checkpoint_no))
{ /* Determine the end of checkpoints in the last log file. */ if (err == DB_SUCCESS)
{
last_checkpoint_no= log_sys.next_checkpoint_no;
ut_ad(last_checkpoint_no > uint16_t(4 * log_sys.is_encrypted()));
ut_ad(last_checkpoint_no <= uint16_t{log_t::START_OFFSET}); if (!recovery_start || i == found_recovery_start) return DB_SUCCESS;
log_sys.next_checkpoint_no= UINT16_MAX; if (const byte *checkpoint_buf= log_sys.checkpoint_buf)
{ /* Clear any garbage that may have been left behind by a crashduringlog_t::write_checkpoint()orlog_t::set_archive().
See also log_t::clear_mmap() and log_t::attach(). */ const size_t bs{log_sys.write_size};
memcpy(last_checkpoint_buf, checkpoint_buf, bs); const size_t tail= (last_checkpoint_no * 8) & (bs - 1);
memset(last_checkpoint_buf + tail, 0, bs - tail);
} goto next;
}
const byte *buf= log_sys.checkpoint_buf; if (!buf)
buf= log_sys.buf; switch (my_betoh32(*reinterpret_cast<const uint32_t*>(buf))) { case log_t::FORMAT_10_8: case log_t::FORMAT_ENC_11: if (recv_check_log_block(buf))
{
was_archive= true; #ifdef HAVE_PMEM if (log_sys.is_mmap_writeable())
log_sys.log.close(); #endif
log_sys.archive= false;
file= OS_FILE_CLOSED; goto circular_log_recovery;
} /* fall through */ goto found_bad_header; default: if (log_file_is_zero()) /* There is no checkpoint in this file. There may be somelogrecords,orthisisazero-filledpreallocated
file. We will know at log_t::set_recovered_lsn(). */; else
found_bad_header:
last_checkpoint_no= uint16_t(4 * log_sys.is_encrypted());
}
} elseif (err == DB_SUCCESS)
{
ut_ad(!recovery_start || i == found_recovery_start); /* Restore the checkpoint number of the last log file. */
log_sys.next_checkpoint_no= last_checkpoint_no;
ut_ad(log_sys.is_mmap() == !log_sys.checkpoint_buf); if (byte *checkpoint_buf= log_sys.checkpoint_buf)
memcpy(checkpoint_buf, last_checkpoint_buf, log_sys.write_size); return DB_SUCCESS;
}
if ((recovery_start ? i == found_recovery_start : read_only) ||
i == start) return err;
next:
log_sys.stash_archive_file();
}
}
ut_ad(!log_sys.archive);
size= os_file_get_size(file); if (!size)
{ if (srv_operation != SRV_OPERATION_NORMAL) goto too_small;
} else
{ if (size < log_t::START_OFFSET + SIZE_OF_FILE_CHECKPOINT)
{
too_small:
sql_print_error("InnoDB: File %s is too small", path.c_str());
err_exit:
os_file_close(file); return DB_ERROR;
} elseif (!log_sys.attach(file, size, log_t::log_access(read_only))) goto err_exit; else
file= OS_FILE_CLOSED;
}
first_lsn= mach_read_from_8(buf + LOG_HEADER_START_LSN);
log_sys.set_first_lsn(first_lsn); char creator[LOG_HEADER_CREATOR_END - LOG_HEADER_CREATOR + 1];
memcpy(creator, buf + LOG_HEADER_CREATOR, sizeof creator); /* Ensure that the string is NUL-terminated. */
creator[LOG_HEADER_CREATOR_END - LOG_HEADER_CREATOR]= 0;
lsn_t lsn_offset= 0;
switch (log_sys.format) { default:
sql_print_error("InnoDB: Unsupported redo log format." " The redo log was created with %s.", creator); return DB_ERROR; case log_t::FORMAT_ENC_11: case log_t::FORMAT_10_8: if (files.size() != 1)
{
sql_print_error("InnoDB: Expecting only ib_logfile0"); return DB_CORRUPTION;
}
if (*reinterpret_cast<const uint32_t*>(buf + LOG_HEADER_FORMAT + 4) ||
first_lsn < log_t::FIRST_LSN)
{
sql_print_error("InnoDB: Invalid ib_logfile0 header block;" " the log was created with %s.", creator); return DB_CORRUPTION;
}
if (!mach_read_from_4(buf + LOG_HEADER_CREATOR_END) &&
log_sys.format == log_t::FORMAT_10_8); elseif (!log_crypt_read_header(buf + LOG_HEADER_CREATOR_END, false))
{
sql_print_error("InnoDB: Reading log encryption info failed;" " the log was created with %s.", creator); return DB_ERROR;
} elseif (log_sys.format == log_t::FORMAT_10_8)
log_sys.format= log_t::FORMAT_ENC_10_8;
if (checkpoint_lsn >= log_sys.last_checkpoint_lsn)
log_sys.set_recovered_checkpoint(checkpoint_lsn, lsn= end_lsn,
field == log_t::CHECKPOINT_1);
} if (!log_sys.last_checkpoint_lsn) goto got_no_checkpoint; elseif (!log_sys.archived_lsn)
log_sys.archived_lsn= lsn; if (recv_sys_invalid_rpo(lsn)) return DB_READ_ONLY; if (!memcmp(creator, "Backup ", 7) &&
!recv_sys.rpo && !high_level_read_only)
srv_start_after_restore= true;
if (!tmp_buf)
{
tmp_buf= static_cast<byte*>
(ut_malloc_dontdump(tmp_buf_size, PSI_INSTRUMENT_ME)); if (!tmp_buf) return DB_OUT_OF_MEMORY;
} return DB_SUCCESS; case log_t::FORMAT_10_5: case log_t::FORMAT_10_5 | log_t::FORMAT_ENCRYPTED: if (files.size() != 1)
{
sql_print_error("InnoDB: Expecting only ib_logfile0"); return DB_CORRUPTION;
} /* fall through */ case log_t::FORMAT_10_2: case log_t::FORMAT_10_2 | log_t::FORMAT_ENCRYPTED: case log_t::FORMAT_10_3: case log_t::FORMAT_10_3 | log_t::FORMAT_ENCRYPTED: case log_t::FORMAT_10_4: case log_t::FORMAT_10_4 | log_t::FORMAT_ENCRYPTED:
uint64_t max_no= 0; const lsn_t log_size{(log_sys.file_size - 2048) * files.size()}; for (size_t field= 512; field < 2048; field += 1024)
{ const byte *b = buf + field;
if (!recv_check_log_block(b))
{
DBUG_PRINT("ib_log", ("invalid checkpoint checksum at %zu", field)); continue;
}
if (log_sys.is_encrypted() && !log_crypt_read_checkpoint_buf(b))
{
sql_print_error("InnoDB: Reading checkpoint encryption info failed."); continue;
}
if (!log_sys.last_checkpoint_lsn)
{
got_no_checkpoint:
sql_print_error("InnoDB: No valid checkpoint was found;" " the log was created with %s.", creator); return DB_ERROR;
}
if (first_lsn == LSN_MAX) return DB_CORRUPTION;
if (dberr_t err= recv_log_recover_10_5(lsn_offset))
{ constchar *msg1, *msg2, *msg3;
msg1= srv_operation == SRV_OPERATION_NORMAL
? "InnoDB: Upgrade after a crash is not supported."
: "mariadb-backup --prepare is not possible.";
if (err == DB_ERROR)
{
msg2= srv_operation == SRV_OPERATION_NORMAL
? ". You must start up and shut down MariaDB "
: ". You must use mariadb-backup ";
msg3= (log_sys.format & ~log_t::FORMAT_ENCRYPTED) == log_t::FORMAT_10_5
? "10.7 or earlier." : "10.4 or earlier.";
} else
msg2= ", and it appears corrupted.", msg3= "";
sql_print_error("%s The redo log was created with %s%s%s",
msg1, creator, msg2, msg3); return err;
}
goto upgrade;
}
/** Trim old log records for a page. @paramstart_lsnoldestlogsequencenumbertopreserve
@return whether all the log for the page was trimmed */ inlinebool page_recv_t::trim(lsn_t start_lsn)
{ while (log.head)
{ if (log.head->lsn > start_lsn) returnfalse;
last_offset= 1; /* the next record must not be same_page */
log_rec_t *next= log.head->next;
recv_sys.free(log.head);
log.head= next;
}
log.tail= nullptr; returntrue;
}
/** Ignore any earlier redo log records for this page. */ inlinevoid page_recv_t::will_not_read()
{
ut_ad(!being_processed);
skip_read= true;
log.clear();
}
if (pages_it != pages.end() && pages_it->second.being_processed < 0)
pages_it= pages.end();
for (map::iterator p= pages.begin(); p != pages.end(); )
{ if (p->second.being_processed < 0)
{
map::iterator r= p++;
erase(r);
} else
p++;
}
}
/** Allocate a block from the buffer pool for recv_sys.pages */
ATTRIBUTE_COLD buf_block_t *recv_sys_t::add_block()
{ for (bool freed= false;;)
{ constauto rs= UT_LIST_GET_LEN(blocks) * 2;
mysql_mutex_lock(&buf_pool.mutex); constauto bs=
UT_LIST_GET_LEN(buf_pool.free) + UT_LIST_GET_LEN(buf_pool.LRU); if (UNIV_LIKELY(bs > BUF_LRU_MIN_LEN || rs < bs))
{
buf_block_t *block= buf_LRU_get_free_block(have_mutex);
mysql_mutex_unlock(&buf_pool.mutex); return block;
} /* out of memory: redo log occupies more than 1/3 of buf_pool
and there are fewer than BUF_LRU_MIN_LEN pages left */
mysql_mutex_unlock(&buf_pool.mutex); if (freed) return nullptr;
freed= true;
garbage_collect();
}
}
/** Wait for buffer pool to become available. */
ATTRIBUTE_COLD void recv_sys_t::wait_for_pool(size_t pages)
{
mysql_mutex_unlock(&mutex);
os_aio_wait_until_no_pending_reads(false);
os_aio_wait_until_no_pending_writes(false);
mysql_mutex_lock(&mutex);
garbage_collect();
mysql_mutex_lock(&buf_pool.mutex); const size_t available= UT_LIST_GET_LEN(buf_pool.free);
mysql_mutex_unlock(&buf_pool.mutex); if (available < pages)
buf_flush_sync_batch(lsn, false);
}
/** Register a redo log snippet for a page. @paramitpageiterator @paramlredologsnippet @paramlenlengthofl,inbytes
@return whether we ran out of memory */
ATTRIBUTE_NOINLINE bool recv_sys_t::add(map::iterator it, const byte *l, size_t len)
{
mysql_mutex_assert_owner(&mutex);
page_recv_t &recs= it->second;
buf_block_t *block;
switch (*l & 0x70) { case FREE_PAGE: case INIT_PAGE:
recs.will_not_read();
mlog_init.add(it->first, start_lsn); /* FIXME: remove this! */ /* fall through */ default:
log_phys_t *tail= static_cast<log_phys_t*>(recs.log.last()); if (!tail) break; if (tail->start_lsn != start_lsn) break;
ut_ad(tail->lsn == lsn);
block= UT_LIST_GET_LAST(blocks);
ut_ad(block); const size_t used= uint16_t(block->page.free_offset - 1) + 1;
ut_ad(used >= ALIGNMENT); const byte *end= const_cast<const log_phys_t*>(tail)->end(); if (!((reinterpret_cast<size_t>(end + len) ^ reinterpret_cast<size_t>(end)) & ~(ALIGNMENT - 1)))
{ /* Use already allocated 'padding' bytes */
append:
MEM_MAKE_ADDRESSABLE(end + 1, len); /* Append to the preceding record for the page */
tail->append(l, len); returnfalse;
} if (end <= &block->page.frame[used - ALIGNMENT] ||
&block->page.frame[used] >= end) break; /* Not the last allocated record in the page */ const size_t new_used= static_cast<size_t>
(end - block->page.frame + len + 1);
ut_ad(new_used > used); if (new_used > srv_page_size) break;
block->page.free_offset=
ut_calc_align<uint16_t>(static_cast<uint16_t>(new_used), ALIGNMENT); goto append;
}
const size_t size{log_phys_t::alloc_size(len)};
ut_ad(size <= srv_page_size); void *buf;
block= UT_LIST_GET_FIRST(blocks); if (UNIV_UNLIKELY(!block))
{
create_block:
block= add_block(); if (UNIV_UNLIKELY(!block)) returntrue;
block->page.used_records= 1;
block->page.free_offset=
ut_calc_align<uint16_t>(static_cast<uint16_t>(size), ALIGNMENT);
static_assert(ut_is_2pow(ALIGNMENT), "ALIGNMENT must be a power of 2");
UT_LIST_ADD_FIRST(blocks, block);
MEM_MAKE_ADDRESSABLE(block->page.frame, size);
MEM_NOACCESS(block->page.frame + size, srv_page_size - size);
buf= block->page.frame;
} else
{
size_t free_offset= block->page.free_offset;
ut_ad(!ut_2pow_remainder(free_offset, ALIGNMENT)); if (UNIV_UNLIKELY(!free_offset))
{
ut_ad(srv_page_size == 65536); goto create_block;
}
ut_ad(free_offset <= srv_page_size);
free_offset+= size;
if (free_offset > srv_page_size) goto create_block;
/** Buffer wrapper for memory-mapped log_sys.archive,
with the capability to warp from log_sys.buf to log_sys.resize_buf */ struct recv_warp : public recv_buf
{
constexpr recv_warp(const byte *ptr) : recv_buf(ptr) {}
/** Parse a mini-transaction and validate the CRC-32C. @paramlthestartofthemini-transaction @paramnoncesizeoftheencryptionnonce(0=unencrypted,8=encrypted) @retvalOKifvalid(lwillbepositionedaftertherecords) @retvalGOT_EOFifthelogiscorrupted(endofthecircularfile)
@retval PREMATURE_EOF if we ran out of buffer */ template<typename source> static ATTRIBUTE_NOINLINE
recv_sys_t::parse_mtr_result log_parse_start(source &l, unsigned nonce)
noexcept
{
ut_ad(nonce == 0 || nonce == 8);
/* Check that the entire mini-transaction is included within the buffer */ if (l.is_eof(0)) return recv_sys_t::PREMATURE_EOF;
if (*l <= 1) /* We should never write an empty mini-transaction. */ return recv_sys_t::GOT_EOF;
const source begin{l}; for (uint32_t total_len= 0, rlen; !l.is_eof(); l+= rlen, total_len+= rlen)
{ if (total_len >= recv_sys_t::MTR_SIZE_MAX) return recv_sys_t::GOT_EOF; if (*l <= 1) goto eom_found;
rlen= *l & 0xf;
++l; if (!rlen)
{ if (l.is_eof(0)) break;
rlen= mlog_decode_varint_length(*l); if (l.is_eof(rlen)) break; const uint32_t addlen= l.decode_varint(); if (UNIV_UNLIKELY(addlen >= recv_sys_t::MTR_SIZE_MAX)) return recv_sys_t::GOT_EOF;
rlen= addlen + 15;
}
}
/* Not the entire mini-transaction was present. */ return recv_sys_t::PREMATURE_EOF;
/* The sequence bit only matters for detecting log file wrap-around,
which is only possible in the innodb_log_archive=OFF format. */ if (!log_sys.archive && *l != log_sys.get_sequence_bit(end_lsn)) return recv_sys_t::GOT_EOF;
if (l.is_eof(5 + nonce)) return recv_sys_t::PREMATURE_EOF;
ATTRIBUTE_NOINLINE void recv_sys_t::parse_init(const page_id_t id) noexcept
{ /* recv_scan_log() may have stored some log for this page before enteringtheskip_the_rest:loop.Suchrecordsmustbediscarded, becausereadinganINIT_PAGEorFREE_PAGErecordimpliesthatthe pagecanberecoveredbasedonlogrecords,withoutreadingitfrom
a data file. */
mlog_init.add(id, start_lsn);
/** Report that recv_sys_t::parse_tail() encountered an unknown log record. @retvalOKifinnodb_force_recoveryisset
@retval GOT_EOF otherwise */
ATTRIBUTE_COLD ATTRIBUTE_NOINLINE static recv_sys_t::parse_mtr_result log_unknown() noexcept
{ if (srv_force_recovery)
{
sql_print_warning("InnoDB: Ignoring unknown log record at LSN " LSN_PF,
recv_sys.lsn); return recv_sys_t::OK;
}
sql_print_error("InnoDB: Unknown log record at LSN " LSN_PF, recv_sys.lsn); return recv_sys.set_corrupt_log();
}
/** Report that recv_sys_t::parse_tail() encountered a corrupted page_id. @retvalOKifinnodb_force_recoveryisset
@retval GOT_EOF otherwise */
ATTRIBUTE_COLD ATTRIBUTE_NOINLINE static recv_sys_t::parse_mtr_result log_page_id_corrupted() noexcept
{ if (srv_force_recovery)
{
sql_print_warning("InnoDB: Ignoring corrupted page identifier at LSN "
LSN_PF, recv_sys.lsn); return recv_sys_t::OK;
}
sql_print_error("InnoDB: Corrupted page identifier at " LSN_PF "; set innodb_force_recovery=1 to ignore the record.",
recv_sys.lsn); return recv_sys.set_corrupt_log();
}
/** Report that recv_sys_t::parse_tail() encountered a malformed log record. @retvalOKifinnodb_force_recoveryisset
@retval GOT_EOF otherwise */
ATTRIBUTE_COLD ATTRIBUTE_NOINLINE static recv_sys_t::parse_mtr_result log_record_corrupted() noexcept
{ if (srv_force_recovery)
{
sql_print_warning("InnoDB: Ignoring malformed log record at LSN "
LSN_PF, recv_sys.lsn); return recv_sys_t::OK;
}
sql_print_error("InnoDB: Malformed log record at LSN " LSN_PF "; set innodb_force_recovery=1 to ignore.", recv_sys.lsn); return recv_sys.set_corrupt_log();
}
/** Parse a file-level log record. @paramidtablespaceidentifier @paramif_existswhethertocheckifthetablespaceexists @parambfirstbyteoflogrecord @paramllogrecordpayload @paramrlenlengthofthepayload @retvalOKifinnodb_force_recoveryisset
@retval GOT_EOF otherwise */
ATTRIBUTE_COLD ATTRIBUTE_NOINLINE static recv_sys_t::parse_mtr_result
log_parse_file(const page_id_t id, bool if_exists, const byte b, const byte *l, uint32_t rlen)
{ if (recv_sys.scanned_lsn > recv_sys.lsn) /* Wemustalreadyhaveparsedthisrecord,intheskip_the_rest: loopinrecv_scan_log()beforestartingmulti-batchrecovery.
*/ return recv_sys_t::OK; if (UNIV_LIKELY(rlen)); elseif (!id.raw() && b == FILE_CHECKPOINT + 2) /* FILE_CHECKPOINT followed by any number of NUL bytes could be
used for padding the log. */ return recv_sys_t::OK; else
file_rec_error: return log_record_corrupted();
if (UNIV_UNLIKELY(id.page_no())) goto file_rec_error;
const uint32_t space_id{id.space()};
switch (b & 0xf0) { default: goto file_rec_error; case FILE_CHECKPOINT: if (UNIV_UNLIKELY(space_id || l[rlen] > 1)) /* FILE_CHECKPOINT must be the last record of a mini-transaction. */ goto file_rec_error; if (UNIV_UNLIKELY(rlen != 8))
{ if (rlen > UNIV_PAGE_SIZE_MAX || memcmp(l, field_ref_zero, rlen)) goto file_rec_error; break;
} if (const lsn_t c= mach_read_from_8(l))
{ if (UNIV_UNLIKELY(srv_print_verbose_log == 2))
fprintf(stderr, "FILE_CHECKPOINT(" LSN_PF ") %s at " LSN_PF "\n",
c, c != log_sys.last_checkpoint_lsn
? "ignored" : recv_sys.file_checkpoint ? "reread" : "read",
recv_sys.lsn);
/* There can be multiple FILE_CHECKPOINT for the same LSN. */ if (!recv_sys.file_checkpoint)
{
ut_ad(log_sys.last_checkpoint_lsn || log_sys.archive); if (!log_sys.last_checkpoint_lsn)
log_sys.last_checkpoint_lsn= c;
recv_sys.file_checkpoint= recv_sys.lsn; return recv_sys_t::GOT_EOF;
}
} break; case FILE_DELETE: case FILE_MODIFY: case FILE_RENAME: case FILE_CREATE: if (!space_id) goto file_rec_error; /* There is no terminating NUL character. Names must end in .ibd.
For FILE_RENAME, there is a NUL between the two file names. */
if (fn2)
{
fn2++; if (memchr(fn2, 0, fn2end - fn2)) goto file_rec_error; if (fn2end - fn2 < 4 || memcmp(fn2end - 4, DOT_IBD, 4)) goto file_rec_error;
}
if (space_id == TRX_SYS_SPACE || srv_is_undo_tablespace(space_id)) goto file_rec_error; if (fnend - l < 4 ||
(memcmp(fnend - 4, DOT_IBD, 4) && memcmp(fnend - 4, DOT_IBB, 4))) goto file_rec_error;
if (UNIV_UNLIKELY(!recv_needed_recovery && srv_read_only_mode)) break;
if (srv_operation == SRV_OPERATION_BACKUP ||
srv_operation == SRV_OPERATION_BACKUP_NO_DEFER)
{ if (log_file_op)
log_file_op(space_id, b & 0xf0, l, fnend - l, fn2,
fn2 ? fn2end - fn2 : 0); break;
}
if (!log_sys.last_checkpoint_lsn)
{ /* We are currently validating checkpoints in recv_log_t::find_checkpoint_archived().Wemustnotopenand validatedatafilesuntilweactuallystartrecoveryfroma checkpoint,becausetherecouldbelotsofFILE_MODIFYand
FILE_CHECKPOINT log records to be parsed. */
ut_ad(!recv_sys.file_checkpoint);
ut_ad(log_sys.archive); return recv_sys_t::OK;
}
if (fn2)
{
fil_name_process(reinterpret_cast<constchar*>(fn2), fn2end - fn2,
space_id, mfile_type_t(b & 0xf0),
recv_sys.start_lsn, if_exists); if (recv_sys.file_checkpoint)
{ constchar *name= reinterpret_cast<constchar*>(fn2); const size_t len= fn2end - fn2; auto r= renamed_spaces.emplace(space_id, std::string{name,len}); if (!r.second)
r.first->second= std::string{name, len};
}
}
if (recv_sys.is_corrupt_fs()) return recv_sys_t::GOT_EOF;
}
return recv_sys_t::OK;
}
/** Register a page-level record. @paramspace_idtablespace @parampage_nopagenumber @retvalOKifnoproblem @retvalPREMATURE_EOFifthelogiscorruptedbuttheerrorisignored
@retval GOT_EOF if the log is corrupted */
ATTRIBUTE_NOINLINE static recv_sys_t::parse_mtr_result
log_page_modify(uint32_t space_id, uint32_t page_no) noexcept
{
ut_ad(space_id); if (space_id >= SRV_SPACE_ID_UPPER_BOUND || srv_is_undo_tablespace(space_id)) return recv_sys_t::OK;
recv_spaces_t::iterator i= recv_spaces.lower_bound(space_id); const lsn_t lsn{recv_sys.lsn}; if (i != recv_spaces.end() && i->first == space_id); elseif (lsn < recv_sys.file_checkpoint) /* We have not seen all records between the checkpoint and FILE_CHECKPOINT.ThereshouldbeaFILE_DELETEorFILE_MODIFY
for this tablespace later, to be handled in fil_name_process(). */
recv_spaces.emplace_hint(i, space_id, file_name_t("", false)); elseif (!srv_read_only_mode)
{ if (!srv_force_recovery)
{
sql_print_error("InnoDB: Missing FILE_DELETE or FILE_MODIFY for " "[page id: space=" UINT32PF ", page number=" UINT32PF "] at " LSN_PF "; set innodb_force_recovery=1 to ignore the record.",
space_id, page_no, lsn); return recv_sys.set_corrupt_log();
}
sql_print_warning("InnoDB: Ignoring record for " "[page id: space=" UINT32PF ", page number=" UINT32PF "] at " LSN_PF,
space_id, page_no, lsn); return recv_sys_t::PREMATURE_EOF;
} return recv_sys_t::OK;
}
ut_d(std::set<page_id_t> freed); #if0 && defined UNIV_DEBUG /* MDEV-21727 FIXME: enable this */ /* Pages that have been modified in this mini-transaction. Ifamini-transactionwritesINIT_PAGEforapage,itshouldnothave writtenanylogrecordsforthepage.Unfortunately,thisdoesnot holdforROW_FORMAT=COMPRESSEDpages,becausepage_zip_compress() canbeinvokedinapessimisticoperation,evenafterloghas
been written for other pages. */
ut_d(std::set<page_id_t> modified); #endif
uint32_t space_id= 0, page_no= 0; /* The end offset the last write (always 0 in storing==BACKUP).
The value 1 means that no "same page" record is allowed. */
uint last_offset= 0; bool got_page_op= false;
if (UNIV_UNLIKELY((b & 0x70) == RESERVED)) if (parse_mtr_result r= log_unknown()) return r;
l= mtr_t::parse_length(l, &rlen); bool is_binlog= false;
uint32_t idlen; if ((b & 0x80) && got_page_op)
{ if (storing == BACKUP) continue; /* This record is for the same page as the previous one. */ if (UNIV_UNLIKELY((b & 0x70) <= INIT_PAGE))
{ /* FREE_PAGE,INIT_PAGE cannot be with same_page flag */
record_corrupted: if (parse_mtr_result r= log_record_corrupted()) return r; /* the next record must not be same_page */
last_offset= 1; continue;
}
DBUG_PRINT("ib_log",
("scan " LSN_PF ": rec %x len %zu page %u:%u",
lsn, b, l - recs + rlen, space_id, page_no)); goto same_page;
} if (storing != BACKUP) last_offset= 0;
idlen= mlog_decode_varint_length(*l); if (UNIV_UNLIKELY(idlen > 5 || idlen >= rlen))
{ if (!*l && b == FILE_CHECKPOINT + 1) continue;
page_id_corrupted: if (parse_mtr_result r= log_page_id_corrupted()) return r; continue;
}
space_id= mlog_decode_varint(l); if (UNIV_UNLIKELY(space_id == MLOG_DECODE_ERROR)) goto page_id_corrupted;
static_assert((LOG_BINLOG_ID_0 | 1) == LOG_BINLOG_ID_1, "");
is_binlog= !ENC_10_8 && (space_id | 1) == LOG_BINLOG_ID_1;
l+= idlen;
rlen-= idlen;
idlen= mlog_decode_varint_length(*l); if (UNIV_UNLIKELY(idlen > 5 || idlen > rlen)) goto page_id_corrupted;
page_no= mlog_decode_varint(l); if (UNIV_UNLIKELY(page_no == MLOG_DECODE_ERROR)) goto page_id_corrupted;
l+= idlen;
rlen-= idlen; if (storing != BACKUP && ENC_10_8)
{
mach_write_to_4(iv + 8, space_id);
mach_write_to_4(iv + 12, page_no);
}
DBUG_PRINT("ib_log",
("scan " LSN_PF ": rec %x len %zu page %u:%u",
lsn, b, l - recs + rlen, space_id, page_no));
got_page_op= !(b & 0x80); if (got_page_op)
{ if (storing == BACKUP)
{ if (page_no == 0 && (b & 0xf0) == INIT_PAGE && first_page_init)
first_page_init(space_id); elseif (rlen == 1 && undo_space_trunc)
{ if (ENC_10_8)
{
mach_write_to_4(iv + 8, space_id);
mach_write_to_4(iv + 12, page_no); if (*log_decrypt_legacy(iv, recs, l, 1, decrypt_buf) != TRIM_PAGES) continue;
} elseif (*l != TRIM_PAGES) continue;
undo_space_trunc(space_id);
} continue;
} if (storing == YES && UNIV_LIKELY(space_id != TRX_SYS_SPACE)) if (parse_mtr_result r= log_page_modify(space_id, page_no))
{ if (UNIV_UNLIKELY(r == PREMATURE_EOF)) continue; return r;
}
same_page: if (!rlen); elseif (!is_binlog && UNIV_UNLIKELY(size_t(l - recs) + rlen > srv_page_size)) goto record_corrupted; const page_id_t id{space_id, page_no};
ut_d(if ((b & 0x70) == INIT_PAGE || (b & 0x70) == OPTION)
freed.erase(id));
ut_ad(freed.find(id) == freed.end()); const byte *cl{ENC_10_8 ? l : nullptr}; switch (b & 0x70) { case FREE_PAGE:
ut_ad(freed.emplace(id).second); /* the next record must not be same_page */ if (storing != BACKUP) last_offset= 1; goto free_or_init_page; case INIT_PAGE: if (storing != BACKUP) last_offset= FIL_PAGE_TYPE;
free_or_init_page: if (UNIV_UNLIKELY(rlen != 0)) goto record_corrupted;
store_freed_or_init_rec(id, (b & 0x70) == FREE_PAGE);
if (storing == NO)
{ /* We must update mlog_init for the correct operation of multi-batchrecovery,forexampletoavoidoccasional failuresofthetestinnodb.recovery_memory.
For storing == YES, add() will invoke mlog_init.add(). */
parse_init(id); continue;
} break; case EXTENDED: if (storing == NO) /* We really only care about WRITE records to page 0, to invokefil_space_set_recv_size_and_flags().Asofnow,the EXTENDEDrecordsrefertoindexorundologpages(which page0nevercanbe),orwehavetheTRIM_PAGESsubtypefor shrinkingatablespace,toalargernumberofpagesthan0. Eitherway,wecanignorethisrecordduringthepreparation
for multi-batch recovery. */ continue; if (UNIV_UNLIKELY(!rlen)) goto record_corrupted; if (ENC_10_8 && (storing == YES || rlen == 1))
cl= log_decrypt_legacy(iv, recs, l, rlen, decrypt_buf);
if (rlen == 1 && *(ENC_10_8 ? cl : l) == TRIM_PAGES)
{ if (srv_is_undo_tablespace(space_id))
{ if (page_no != SRV_UNDO_TABLESPACE_SIZE_IN_PAGES) goto record_corrupted; /* The entire undo tablespace will be reinitialized by innodb_undo_log_truncate=ON.Discardoldlogforall
pages. */
trim({space_id, 0}, start_lsn);
truncated_undo_spaces[space_id - srv_undo_space_id_start]=
{ start_lsn, page_no};
} elseif (space_id != 0) goto record_corrupted; else
{ /* Shrink the system tablespace */
trim({space_id, page_no}, start_lsn);
truncated_sys_space= {start_lsn, page_no};
}
static_assert(UT_ARR_SIZE(truncated_undo_spaces) ==
TRX_SYS_MAX_UNDO_SPACES, "compatibility"); /* the next record must not be same_page */ if (storing != BACKUP) last_offset= 1; continue;
} /* This record applies to an undo log or index page, and it maybefollowedbysubsequentWRITEorsimilarrecordsforthe
same page in the same mini-transaction. */ if (storing != BACKUP) last_offset= FIL_PAGE_TYPE; break; case OPTION: /* OPTION records can be safely ignored in recovery */ if (storing == YES &&
rlen == 5/* OPT_PAGE_CHECKSUM and CRC-32C; see page_checksum() */)
{ if (ENC_10_8)
cl= log_decrypt_legacy(iv, recs, l, rlen, decrypt_buf); if (*(ENC_10_8 ? cl : l) == OPT_PAGE_CHECKSUM) break;
} /* fall through */ case RESERVED: continue; case MEMMOVE: case MEMSET: if (storing == YES && !ENC_10_8 && is_binlog) goto record_corrupted; /* fall through */ case WRITE: if (storing == BACKUP) continue; if (storing == NO && UNIV_LIKELY((page_no | (uint32_t)is_binlog) != 0)) /* fil_space_set_recv_size_and_flags() is mandatory for storing==NO. Itisonlyapplicabletopage_no==0.Otherthanthat,wecanjust ignorethepayloadandonlycomputethemini-transactionchecksum;
there will be a subsequent call with storing==YES. */ continue; if (UNIV_UNLIKELY(rlen == 0 || last_offset == 1)) goto record_corrupted; if (ENC_10_8)
cl= log_decrypt_legacy(iv, recs, l, rlen, decrypt_buf); if (!is_binlog)
{ const uint32_t olen= mlog_decode_varint_length(*(ENC_10_8 ? cl : l)); if (UNIV_UNLIKELY(olen >= rlen) || UNIV_UNLIKELY(olen > 3)) goto record_corrupted; const uint32_t offset= mlog_decode_varint(ENC_10_8 ? cl : l);
ut_ad(offset != MLOG_DECODE_ERROR);
static_assert(FIL_PAGE_OFFSET == 4, "compatibility"); if (UNIV_UNLIKELY(offset >= srv_page_size)) goto record_corrupted;
last_offset+= offset; if (UNIV_UNLIKELY(last_offset < 8 || last_offset >= srv_page_size)) goto record_corrupted;
(ENC_10_8 ? cl : l)+= olen;
rlen-= olen;
} if ((b & 0x70) == WRITE)
{ if (!ENC_10_8 && is_binlog); elseif (UNIV_UNLIKELY(rlen + last_offset > srv_page_size)) goto record_corrupted; elseif (UNIV_UNLIKELY(!page_no) && file_checkpoint)
{ constbool has_size= last_offset <= FSP_HEADER_OFFSET + FSP_SIZE &&
last_offset + rlen >= FSP_HEADER_OFFSET + FSP_SIZE + 4; constbool has_flags= last_offset <=
FSP_HEADER_OFFSET + FSP_SPACE_FLAGS &&
last_offset + rlen >= FSP_HEADER_OFFSET + FSP_SPACE_FLAGS + 4;
ut_ad(storing == NO || space_id == TRX_SYS_SPACE ||
!(has_size || has_flags) ||
recv_spaces.find(space_id) != recv_spaces.end() ||
srv_is_undo_tablespace(space_id));
parse_page0(id, (ENC_10_8 ? cl : l) - last_offset,
has_size, has_flags);
}
parsed_ok:
last_offset+= rlen; if (ENC_10_8 && size_t(cl - decrypt_buf) < srv_page_size)
(l= recs)+= cl - decrypt_buf; break;
}
uint32_t llen= mlog_decode_varint_length(*(ENC_10_8 ? cl : l)); if (UNIV_UNLIKELY(llen > rlen || llen > 3)) goto record_corrupted; const uint32_t len= mlog_decode_varint(ENC_10_8 ? cl : l);
ut_ad(len != MLOG_DECODE_ERROR); if (UNIV_UNLIKELY(last_offset + len > srv_page_size) || is_binlog) goto record_corrupted;
(ENC_10_8 ? cl : l)+= llen;
rlen-= llen;
llen= len; if ((b & 0x70) == MEMSET)
{ if (UNIV_UNLIKELY(rlen > llen)) goto record_corrupted; goto parsed_ok;
} const uint32_t slen= mlog_decode_varint_length(*(ENC_10_8 ? cl : l)); if (UNIV_UNLIKELY(slen != rlen || slen > 3)) goto record_corrupted;
uint32_t s= mlog_decode_varint(ENC_10_8 ? cl : l);
ut_ad(slen != MLOG_DECODE_ERROR); if (s & 1)
s= last_offset - (s >> 1) - 1; else
s= last_offset + (s >> 1) + 1; if (UNIV_UNLIKELY(s < 8 || s + llen > srv_page_size)) goto record_corrupted; goto parsed_ok;
} #if0 && defined UNIV_DEBUG switch (b & 0x70) { case RESERVED:
ut_ad(0); /* we did "continue" earlier */ break; case OPTION: case FREE_PAGE: break; default:
ut_ad(modified.emplace(id).second || (b & 0x70) != INIT_PAGE);
} #endif if (storing != YES); elseif (!ENC_10_8 && is_binlog)
{ if (parse_store_binlog(space_id, l, rlen, page_no, start_lsn, lsn)) goto record_corrupted;
} elseif (if_exists && !parse_store_if_exists(space_id)); elseif (UNIV_UNLIKELY(parse_store(id, ENC_10_8 && l != cl
? decrypt_buf : recs,
(l + rlen) - recs)))
{
rewind(begin, l + rlen); if (!if_exists) return parse_oom();
apply(false); if (is_corrupt_fs()) return GOT_EOF; goto restart;
}
} elseif (parse_mtr_result r= log_parse_file(page_id_t{space_id, page_no},
if_exists, b, l, rlen)) return r;
}
ut_ad(l + log_sys.is_encrypted() * 8 + 5 == el);
bool log_t::archived_switch_recovery() noexcept
{ if (!archived_switch_recovery_prepare(first_lsn + capacity())) returnfalse;
if (uint16_t(~next_checkpoint_no))
{ /* We have completed recv_sys_t::find_checkpoint_archived(),
and archived_switch_recovery_rewind_checkpoint() will not be called. */ if (is_mmap())
{
my_munmap(buf, size_t(file_size));
buf= nullptr;
}
log.close();
}
template<recv_sys_t::store storing,uint32_t format>
recv_sys_t::parse_mtr_result recv_sys_t::parse_mmap(bool if_exists)
{
recv_sys_t::parse_mtr_result r{parse_mtr<storing,format>(if_exists)}; if (UNIV_LIKELY(r != PREMATURE_EOF) || !log_sys.is_mmap()) return r;
ut_ad(recv_sys.len == log_sys.file_size);
ut_ad(recv_sys.offset >= log_sys.START_OFFSET);
ut_ad(recv_sys.offset <= recv_sys.len); if (log_sys.archive)
{ if (!log_sys.archived_mmap_switch())
{
ut_d(auto i= recv_sys.log_archive.find(log_sys.get_first_lsn()));
ut_ad(i != recv_sys.log_archive.end());
ut_d(i++);
ut_ad(i == recv_sys.log_archive.end()); return GOT_EOF;
}
recv_warp s{&log_sys.buf[recv_sys.offset]}; auto r= recv_sys.parse<recv_warp,storing,format>(s,if_exists); if (UNIV_LIKELY(r != GOT_OOM))
log_sys.archived_mmap_switch_recovery_complete(); else /* rewind() must have closed the next file when invoking
log_sys.set_recovered_lsn(start_lsn) */
ut_ad(!log_sys.archived_mmap_switch()); return r;
}
recv_ring s
{recv_sys.offset == recv_sys.len
? &log_sys.buf[log_sys.START_OFFSET]
: &log_sys.buf[recv_sys.offset]}; return recv_sys.parse<recv_ring,storing,format>(s,if_exists);
}
/* The log_t::format::ENC_10_8 is needed for a rare case of recovering or
upgrading from an old backup or database with innodb_encrypt_log=ON. */ template ATTRIBUTE_COLD
recv_sys_t::parse_mtr_result recv_sys_t::parse_mtr
<recv_sys_t::store::NO,log_t::FORMAT_ENC_10_8>(bool); template ATTRIBUTE_COLD
recv_sys_t::parse_mtr_result recv_sys_t::parse_mtr
<recv_sys_t::store::YES,log_t::FORMAT_ENC_10_8>(bool); template ATTRIBUTE_COLD
recv_sys_t::parse_mtr_result recv_sys_t::parse_mtr
<recv_sys_t::store::BACKUP,log_t::FORMAT_ENC_10_8>(bool);
/** @return the parsing function for mariadb-backup --backup */
recv_sys_t::parser recv_sys_t::get_backup_parser() noexcept
{ if (log_sys.is_mmap()) switch (log_sys.format) { case log_t::FORMAT_10_8: return parse_mmap<BACKUP,log_t::FORMAT_10_8>; case log_t::FORMAT_ENC_11: return parse_mmap<BACKUP,log_t::FORMAT_ENC_11>; case log_t::FORMAT_ENC_10_8: return parse_mmap<BACKUP,log_t::FORMAT_ENC_10_8>;
} else switch (log_sys.format) { case log_t::FORMAT_10_8: return parse_mtr<BACKUP,log_t::FORMAT_10_8>; case log_t::FORMAT_ENC_11: return parse_mtr<BACKUP,log_t::FORMAT_ENC_11>; case log_t::FORMAT_ENC_10_8: return parse_mtr<BACKUP,log_t::FORMAT_ENC_10_8>;
}
ut_error;
}
/** Apply the hashed log records to the page, if the page lsn is less than the lsnofalogrecord. @param[in,out]blockbufferpoolpage @param[in,out]mtrmini-transaction @param[in,out]recslogrecordstoapply @param[in,out]spacetablespace,orNULLifnotlookedupyet @param[in,out]init_lsnpageinitializationLSN,or0 @returntherecoveredpage
@retval nullptr on failure */ static buf_block_t *recv_recover_page(buf_block_t *block, mtr_t &mtr,
page_recv_t &recs,
fil_space_t *space, lsn_t init_lsn = 0)
{
mysql_mutex_assert_not_owner(&recv_sys.mutex);
ut_ad(recv_sys.apply_log_recs);
ut_ad(recv_needed_recovery);
ut_ad(recs.being_processed == 1);
ut_ad(!space || space->id == block->page.id().space());
ut_ad(log_sys.is_recoverable());
if (UNIV_UNLIKELY(srv_print_verbose_log == 2)) {
ib::info() << "Applying log to page " << block->page.id();
}
DBUG_PRINT("ib_log", ("Applying log to page %u:%u",
block->page.id().space(),
block->page.id().page_no()));
/* There is no need to check LSN for just initialized pages. */ if (skipped_after_init) {
skipped_after_init = false;
ut_ad(end_lsn == page_lsn); if (end_lsn != page_lsn) {
sql_print_information( "InnoDB: The last skipped log record" " LSN " LSN_PF " is not equal to page LSN " LSN_PF,
end_lsn, page_lsn);
}
}
switch (a) { case log_phys_t::APPLIED_NO:
ut_ad(!mtr.has_modifications());
free_page = true;
start_lsn = 0; continue; case log_phys_t::APPLIED_YES: case log_phys_t::APPLIED_CORRUPTED: goto set_start_lsn; case log_phys_t::APPLIED_TO_FSP_HEADER: case log_phys_t::APPLIED_TO_ENCRYPTION: break;
}
if (start_lsn) {
ut_ad(end_lsn >= start_lsn);
ut_ad(!block->page.oldest_modification());
mach_write_to_8(FIL_PAGE_LSN + frame, end_lsn); if (UNIV_LIKELY(!block->page.zip.data)) {
mach_write_to_8(srv_page_size
- FIL_PAGE_END_LSN_OLD_CHKSUM
+ frame, end_lsn);
} else {
buf_zip_decompress(block, false);
} /* The following is adapted from
buf_pool_t::insert_into_flush_list() */
mysql_mutex_lock(&buf_pool.flush_list_mutex);
buf_pool.flush_list_bytes+= block->physical_size();
block->page.set_oldest_modification(start_lsn);
UT_LIST_ADD_FIRST(buf_pool.flush_list, &block->page);
buf_pool.page_cleaner_wakeup();
mysql_mutex_unlock(&buf_pool.flush_list_mutex);
} elseif (free_page && init_lsn) { /* There have been no operations that modify the page.
Any buffered changes will be merged in ibuf_upgrade(). */
ut_ad(!mtr.has_modifications());
block->page.set_freed(block->page.state());
}
/* Make sure that committing mtr does not change the modification
lsn values of page */
mtr.discard_modifications();
mtr.commit();
return block;
}
/** Remove records for a corrupted page. @parampage_idcorruptedpageidentifier @paramnodefileforwhichanerroristobereported
@return whether an error message was reported */
ATTRIBUTE_COLD bool recv_sys_t::free_corrupted_page(page_id_t page_id, const fil_node_t &node) noexcept
{ if (!recovery_on) returnfalse;
p->second.being_processed= -1; if (!srv_force_recovery)
set_corrupt_fs();
mysql_mutex_unlock(&mutex);
(srv_force_recovery ? sql_print_warning : sql_print_error)
("InnoDB: Unable to apply log to corrupted page " UINT32PF " in file %s", page_id.page_no(), node.name); returntrue;
}
ATTRIBUTE_COLD void recv_sys_t::set_corrupt_fs() noexcept
{
mysql_mutex_assert_owner(&mutex); if (!srv_force_recovery)
sql_print_information("InnoDB: Set innodb_force_recovery=1" " to ignore corrupted pages.");
found_corrupt_fs= true;
}
/** Apply any buffered redo log to a page. @paramspacetablespace @parambpagebufferpoolpage
@return whether the page was recovered correctly */ bool recv_recover_page(fil_space_t* space, buf_page_t* bpage)
{
mtr_t mtr{nullptr};
mtr.start();
mtr.set_log_mode(MTR_LOG_NO_REDO);
ut_ad(bpage->frame); /* Move the ownership of the x-latch on the page to this OS thread, sothatwecanacquireasecondx-latchonit.Thisisneededfor
the operations to the page to pass the debug checks. */
bpage->lock.claim_ownership();
bpage->lock.x_lock_recursive();
bpage->fix_on_recovery();
mtr.memo_push(reinterpret_cast<buf_block_t*>(bpage), MTR_MEMO_PAGE_X_FIX);
ut_ad(bpage->frame); /* Move the ownership of the x-latch on the page to this OS thread, sothatwecanacquireasecondx-latchonit.Thisisneededfor
the operations to the page to pass the debug checks. */
bpage->lock.claim_ownership();
bpage->lock.x_lock_recursive();
bpage->fix_on_recovery();
mtr.memo_push(reinterpret_cast<buf_block_t*>(bpage), MTR_MEMO_PAGE_X_FIX);
if (UNIV_UNLIKELY(block != b))
{ /* The page happened to exist in the buffer pool, or it wasjustbeingreadin.Beforetheexclusivepagelatchwasacquiredby
buf_page_create(), all changes to the page must have been applied. */
ut_d(mysql_mutex_lock(&mutex));
ut_ad(pages.find(p->first) == pages.end());
ut_d(mysql_mutex_unlock(&mutex));
space->release(); goto nothing_recoverable;
}
}
/* Released in buf_pool_t::corrupted_evict(), recover_deferred() or below */
block->page.lock.x_lock_recursive();
ut_d(mysql_mutex_lock(&mutex));
ut_ad(&recs == &pages.find(p->first)->second);
ut_d(mysql_mutex_unlock(&mutex));
block= recv_recover_page(block, mtr, recs, space, init_lsn);
ut_ad(mtr.has_committed());
if (space)
{
space->release(); if (block)
block->page.lock.x_unlock();
} return block ? block : reinterpret_cast<buf_block_t*>(-1);
}
/** Read a page or recover it based on redo log records. @parampage_idpageidentifier @parammtrmini-transaction @paramerrerrorcode @returntherequestedblock
@retval nullptr if the page cannot be accessed due to corruption */
ATTRIBUTE_COLD
buf_block_t *
recv_sys_t::recover(const page_id_t page_id, mtr_t *mtr, dberr_t *err)
{ if (!recovery_on)
must_read: return buf_page_get_gen(page_id, 0, RW_NO_LATCH, nullptr, BUF_GET_RECOVER,
mtr, err);
ut_ad(block == free_block); auto s= block->page.fix();
ut_ad(s >= buf_page_t::FREED); /* The block may be write-fixed at this point because we are not
holding a latch, but it must not be read-fixed. */
ut_ad(s < buf_page_t::READ_FIX || s >= buf_page_t::WRITE_FIX); if (s < buf_page_t::UNFIXED)
{
mysql_mutex_lock(&buf_pool.mutex);
block->page.unfix();
buf_LRU_free_page(&block->page, true);
mysql_mutex_unlock(&buf_pool.mutex); goto corrupted;
}
/** Thread-safe function which sorts flush_list by oldest_modification */ staticvoid log_sort_flush_list() noexcept
{ /* Ensure that oldest_modification() cannot change during std::sort() */
{ constdouble pct_lwm= srv_max_dirty_pages_pct_lwm; /* Disable "idle" flushing in order to minimize the wait time below. */
srv_max_dirty_pages_pct_lwm= 0.0;
for (;;)
{
os_aio_wait_until_no_pending_writes(false);
mysql_mutex_lock(&buf_pool.flush_list_mutex); if (buf_pool.page_cleaner_active())
my_cond_wait(&buf_pool.done_flush_list,
&buf_pool.flush_list_mutex.m_mutex); elseif (!os_aio_pending_writes()) break;
mysql_mutex_unlock(&buf_pool.flush_list_mutex);
}
/** Invalidate all pages in the buffer pool.
All pages must be replaceable (not modified, latched, or io-fixed). */
ATTRIBUTE_COLD staticvoid buf_pool_invalidate() noexcept
{
mysql_mutex_lock(&buf_pool.mutex);
ut_ad(!os_aio_pending_reads()); /* os_aio_pending_writes() may hold here if some write_io_callback() didnotreleasetheslotyet.However,buf_flush_sync_batch()waited
for the page write itself to complete, which we will check below. */
ut_d(buf_pool.assert_all_freed());
while (UT_LIST_GET_LEN(buf_pool.LRU))
buf_LRU_scan_and_free_block();
/** Apply buffered log to persistent data pages.
@param last_batch whether it is possible to write more redo log */ void recv_sys_t::apply(bool last_batch)
{
ut_ad(srv_operation <= SRV_OPERATION_EXPORT_RESTORED ||
srv_operation == SRV_OPERATION_RESTORE ||
srv_operation == SRV_OPERATION_RESTORE_EXPORT);
mysql_mutex_assert_owner(&mutex);
garbage_collect();
if (truncated_sys_space.lsn)
{
trim({0, truncated_sys_space.pages}, truncated_sys_space.lsn);
fil_node_t *file= UT_LIST_GET_LAST(fil_system.sys_space->chain);
ut_ad(file->is_open());
/* Last file new size after truncation */
uint32_t new_last_file_size=
truncated_sys_space.pages -
(srv_sys_space.get_min_size()
- srv_sys_space.m_files.at(
srv_sys_space.m_files.size() - 1). param_size());
for (pages_it= pages.begin(); pages_it != pages.end();
pages_it= pages.begin())
{ if (!free_block)
{ if (!last_batch)
log_sys.latch.wr_unlock();
wait_for_pool(1);
pages_it= pages.begin();
mysql_mutex_unlock(&mutex); /* We must release log_sys.latch and recv_sys.mutex before invokingbuf_LRU_get_free_block().Allocatingablockmayinitiate aredologwriteandthereforeacquirelog_sys.latch.Toavoid deadlocks,log_sys.latchmustnotbeacquiredwhileholding
recv_sys.mutex. */
free_block= buf_LRU_get_free_block(have_no_mutex); if (!last_batch)
log_sys.latch.wr_lock();
mysql_mutex_lock(&mutex);
pages_it= pages.begin();
}
while (pages_it != pages.end())
{ if (is_corrupt_fs() || is_corrupt_log())
{ if (space)
space->release(); if (free_block)
{
mysql_mutex_unlock(&mutex);
mysql_mutex_lock(&buf_pool.mutex);
buf_LRU_block_free_non_file_page(free_block);
mysql_mutex_unlock(&buf_pool.mutex);
mysql_mutex_lock(&mutex);
} return;
} if (apply_batch(space_id, space, free_block, last_batch)) break;
}
}
if (space)
space->release();
if (free_block)
{
mysql_mutex_lock(&buf_pool.mutex);
buf_LRU_block_free_non_file_page(free_block);
mysql_mutex_unlock(&buf_pool.mutex);
}
}
if (last_batch)
{
mlog_init.clear();
dblwr.pages.clear();
} else
log_sys.latch.wr_unlock();
mysql_mutex_unlock(&mutex);
if (!last_batch)
{
buf_flush_sync_batch(lsn, false);
buf_pool_invalidate();
log_sys.latch.wr_lock();
} elseif (srv_operation == SRV_OPERATION_RESTORE ||
srv_operation == SRV_OPERATION_RESTORE_EXPORT)
buf_flush_sync_batch(lsn, false); else /* Instead of flushing, last_batch sorts the buf_pool.flush_list
in ascending order of buf_page_t::oldest_modification. */
log_sort_flush_list();
mysql_mutex_lock(&mutex);
ut_d(after_apply= true);
clear();
}
/** Find the end of the circular log when multi-batch recovery is needed. */ static ATTRIBUTE_COLD recv_sys_t::parse_mtr_result
recv_scan_log_circular_skip_the_rest(const recv_sys_t::parser &parser)
{
ut_ad(!log_sys.archive);
recv_sys_t::parse_mtr_result r; while ((r= parser(false)) == recv_sys_t::OK); return r;
}
/** Find the end of the archived log when multi-batch recovery is needed. */ static ATTRIBUTE_COLD recv_sys_t::parse_mtr_result
recv_scan_log_archive_skip_the_rest(const recv_sys_t::parser &parser)
{
ut_ad(log_sys.archive);
recv_sys_t::parse_mtr_result r;
lsn_t first_lsn{log_sys.get_first_lsn()};
log_sys.archived_switch_recovery_prepare(first_lsn + log_sys.capacity());
while ((r= parser(false)) == recv_sys_t::OK)
{ const lsn_t new_first_lsn{log_sys.get_first_lsn()}; if (UNIV_UNLIKELY(first_lsn != new_first_lsn) &&
log_sys.archived_switch_recovery_prepare(new_first_lsn +
log_sys.capacity()))
first_lsn= new_first_lsn;
}
/** Report progress when recovery is taking a long time. */ static ATTRIBUTE_COLD void recv_report_progress()
{ if (recv_sys.report(time(nullptr)))
{ const size_t n= recv_sys.pages.size();
sql_print_information("InnoDB: Parsed redo log up to LSN=" LSN_PF "; to recover: %zu pages", recv_sys.lsn, n);
service_manager_extend_timeout(INNODB_EXTEND_TIMEOUT_INTERVAL, "Parsed redo log up to LSN=" LSN_PF "; to recover: %zu pages", recv_sys.lsn, n);
}
}
/** Parse and store records in a circular log file. */ static ATTRIBUTE_COLD recv_sys_t::parse_mtr_result
recv_scan_circular_store(const recv_sys_t::parser &parser, bool last_phase)
{
ut_ad(!log_sys.archive);
recv_sys_t::parse_mtr_result r;
uint16_t count= 0; while ((r= parser(last_phase)) == recv_sys_t::OK) if (!++count)
recv_report_progress(); return r;
}
/** Parse and store records in the archived log. */ static ATTRIBUTE_COLD recv_sys_t::parse_mtr_result
recv_scan_archive_store(const recv_sys_t::parser &parser, bool last_phase)
{
ut_ad(log_sys.archive);
lsn_t first_lsn{log_sys.get_first_lsn()};
log_sys.archived_switch_recovery_prepare(first_lsn + log_sys.capacity());
recv_sys_t::parse_mtr_result r;
uint16_t count= 0; while ((r= parser(last_phase)) == recv_sys_t::OK)
{ const lsn_t new_first_lsn{log_sys.get_first_lsn()}; if (UNIV_UNLIKELY(first_lsn != new_first_lsn) &&
log_sys.archived_switch_recovery_prepare(new_first_lsn +
log_sys.capacity()))
first_lsn= new_first_lsn; if (!++count)
recv_report_progress();
} return r;
}
/** Scan log and store records to the parsing buffer. @paramlast_phasewhetherchangescanbeappliedtothetablespaces @paramparserlogparsersforstore::NOandstore::YES
@return whether rescan is needed (not everything was stored) */ staticbool
recv_scan_log(constbool last_phase, const recv_sys_t::parser *parser) noexcept
{
DBUG_ENTER("recv_scan_log");
if (recv_sys.report(time(nullptr)))
{
sql_print_information("InnoDB: Read redo log up to LSN=" LSN_PF,
recv_sys.lsn);
service_manager_extend_timeout(INNODB_EXTEND_TIMEOUT_INTERVAL, "Read redo log up to LSN=" LSN_PF,
recv_sys.lsn);
}
recv_sys_t::parse_mtr_result r;
if (UNIV_UNLIKELY(!recv_needed_recovery))
{
ut_ad(!last_phase);
ut_ad(recv_sys.lsn >= log_sys.last_checkpoint_lsn);
if (!store)
{
ut_ad(!recv_sys.file_checkpoint); for (;;)
{ const byte b{log_sys.buf[recv_sys.offset]};
r= parser[false](false); switch (r) { case recv_sys_t::PREMATURE_EOF: goto read_more; default:
ut_ad(r == recv_sys_t::GOT_EOF); break; case recv_sys_t::OK: if (b == FILE_CHECKPOINT + 2 + 8 || (b & 0xf0) == FILE_MODIFY) continue;
}
/** Report a missing tablespace for which page-redo log exists. @param[in]errpreviouserrorcode @param[in]itablespacedescriptor
@return new error code */ static
dberr_t
recv_init_missing_space(dberr_t err, const recv_spaces_t::const_iterator& i)
{ switch (srv_operation) { default: break; case SRV_OPERATION_RESTORE: case SRV_OPERATION_RESTORE_EXPORT: if (i->second.name.find("/#sql") == std::string::npos) {
sql_print_warning("InnoDB: Tablespace " UINT32PF " was not found at %.*s when" " restoring a (partial?) backup." " All redo log" " for this file will be ignored!",
i->first, int(i->second.name.size()),
i->second.name.data());
} return(err);
}
if (srv_force_recovery == 0) {
sql_print_error("InnoDB: Tablespace " UINT32PF " was not" " found at %.*s.", i->first, int(i->second.name.size()),
i->second.name.data());
if (err == DB_SUCCESS) {
sql_print_information( "InnoDB: Set innodb_force_recovery=1 to" " ignore this and to permanently lose" " all changes to the tablespace.");
err = DB_TABLESPACE_NOT_FOUND;
}
} else {
sql_print_warning("InnoDB: Tablespace " UINT32PF " was not found at %.*s" ", and innodb_force_recovery was set." " All redo log for this tablespace" " will be ignored!",
i->first, int(i->second.name.size()),
i->second.name.data());
}
return(err);
}
/** Report the missing tablespace and discard the redo logs for the deleted tablespace. @param[in]rescanrescanofredologsisneeded ifhashtableranoutofmemory @param[out]missing_tablespacemissingtablespaceexistsornot
@return error code or DB_SUCCESS. */ static MY_ATTRIBUTE((warn_unused_result))
dberr_t
recv_validate_tablespace(bool rescan, bool& missing_tablespace)
{
dberr_t err = DB_SUCCESS;
mysql_mutex_lock(&recv_sys.mutex);
for (recv_sys_t::map::iterator p = recv_sys.pages.begin();
p != recv_sys.pages.end();) {
ut_ad(!p->second.log.empty()); const uint32_t space = p->first.space(); if (space == TRX_SYS_SPACE || srv_is_undo_tablespace(space)) {
next:
p++; continue;
}
recv_spaces_t::iterator i = recv_spaces.find(space);
ut_ad(i != recv_spaces.end());
if (deferred_spaces.find(space)) { /* Skip redo logs belonging to
incomplete tablespaces */ goto next;
}
switch (i->second.status) { case file_name_t::NORMAL: goto next; case file_name_t::MISSING: if (srv_operation != SRV_OPERATION_NORMAL) {
} elseif (const lsn_t c = i->second.create_lsn) {
deferred_spaces.add(space, i->second.name, c); goto next;
}
err = recv_init_missing_space(err, i);
i->second.status = file_name_t::DELETED; /* fall through */ case file_name_t::DELETED:
recv_sys_t::map::iterator r = p++;
recv_sys.pages_it_invalidate(r);
recv_sys.erase(r); continue;
}
ut_ad(0);
}
if (err != DB_SUCCESS) {
func_exit:
mysql_mutex_unlock(&recv_sys.mutex); return(err);
}
/* When rescan is not needed, recv_sys.pages will contain the entireredolog.Ifrescanisneededorinnodb_force_recovery
is set, we can ignore missing tablespaces. */ for (const recv_spaces_t::value_type& rs : recv_spaces) { if (UNIV_LIKELY(rs.second.status != file_name_t::MISSING)) { continue;
}
if (deferred_spaces.find(rs.first)) { continue;
}
if (srv_force_recovery) {
sql_print_warning("InnoDB: Tablespace " UINT32PF " was not found at %.*s," " and innodb_force_recovery was set." " All redo log for this tablespace" " will be ignored!",
rs.first, int(rs.second.name.size()),
rs.second.name.data()); continue;
}
if (!rescan) {
sql_print_information("InnoDB: Tablespace " UINT32PF " was not found at '%.*s'," " but there were" " no modifications either.",
rs.first, int(rs.second.name.size()),
rs.second.name.data());
} else {
missing_tablespace = true;
}
}
goto func_exit;
}
/** Check if all tablespaces were found for crash recovery. @param[in]rescanrescanofredologsisneeded @param[out]missing_tablespacemissingtableexists
@return error code or DB_SUCCESS */ static MY_ATTRIBUTE((warn_unused_result))
dberr_t
recv_init_crash_recovery_spaces(bool rescan, bool& missing_tablespace)
{ bool flag_deleted = false;
if (rs.second.status == file_name_t::DELETED) { /* The tablespace was deleted,
so we can ignore any redo log for it. */
flag_deleted = true;
} elseif (rs.second.space != NULL) { /* The tablespace was found, and there
are some redo log records for it. */
fil_names_dirty(rs.second.space);
/* Add the freed page ranges in the respective
tablespace */ if (!rs.second.freed_ranges.empty()
&& (srv_immediate_scrub_data_uncompressed
|| rs.second.space->is_compressed())) {
/** During recovery, rename an archived log file to ib_logfile0. */
ATTRIBUTE_COLD bool log_t::archive_rename() noexcept
{
ut_ad(!archive);
ut_ad(!srv_read_only_mode); /* Only rename if we successfully applied all the log. */ bool success= recv_sys.rpo && recv_sys.rpo < get_flushed_lsn(); if (!success)
{
{ const std::string old_name{get_archive_path(get_first_lsn())};
sql_print_warning("InnoDB: Renaming %.*s to ib_logfile0", int(old_name.size()), old_name.data()); #ifdef _WIN32 /* On Windows, open files cannot be renamed. */
ut_ad(!is_mmap());
log.close(); #endif
success= !rename(old_name);
} #ifdef _WIN32 if (success)
{
log.m_file= os_file_create_func(get_path().c_str(), OS_FILE_OPEN,
OS_LOG_FILE, false, &success);
ut_ad(success == log.is_opened());
} #endif
} return success;
}
inlinebool recv_sys_t::validate_checkpoint() const noexcept
{ const lsn_t last_checkpoint_lsn{log_sys.last_checkpoint_lsn}; if (lsn >= file_checkpoint && lsn >= last_checkpoint_lsn) returnfalse;
sql_print_error("InnoDB: The log was only scanned up to "
LSN_PF ", while the current LSN at the " "time of the latest checkpoint " LSN_PF " was " LSN_PF "!",
lsn, last_checkpoint_lsn, file_checkpoint); returntrue;
}
/** Get the log parser. @tparamstoringNOorYES
@return the recv_sys_t::parse_mmap() for the current log_sys.format */ template<recv_sys_t::store storing> static recv_sys_t::parser get_parse_mmap() noexcept
{
static_assert(storing == recv_sys_t::store::NO ||
storing == recv_sys_t::store::YES, ""); switch (log_sys.format) { case log_t::FORMAT_10_8: return recv_sys_t::parse_mmap<storing, log_t::FORMAT_10_8>; case log_t::FORMAT_ENC_10_8: return recv_sys_t::parse_mmap<storing, log_t::FORMAT_ENC_10_8>; case log_t::FORMAT_ENC_11: return recv_sys_t::parse_mmap<storing, log_t::FORMAT_ENC_11>;
}
ut_error;
}
dberr_t recv_sys_t::find_checkpoint_archived(lsn_t first_lsn, bool silent)
{
ut_ad(log_sys.archive);
ut_ad(!log_sys.checkpoint_buf == log_sys.is_mmap());
ut_ad(!uint16_t(~log_sys.next_checkpoint_no)); const byte *buf; if (byte *c= log_sys.checkpoint_buf)
{
buf= c; if (dberr_t err= log_sys.log.read(0, {c, log_sys.write_size})) return err;
} else
buf= log_sys.buf;
uint16_t n_checkpoint= 0;
{ const uint64_t format{my_betoh64(*reinterpret_cast<const uint64_t*>(buf))}; if (srv_encrypt_log ? format != 1 : format < log_t::START_OFFSET)
{ /* If we are able to start recovery from a previous file and recoveruptotheendofthelastfile,thesubsequentcallto
log_t::write_checkpoint() will fix the last file header. */ if (!silent)
sql_print_error((format == 1 || format >= log_t::START_OFFSET)
? "InnoDB: " LOG_ARCHIVE_NAME " does not match innodb_encrypt_log"
: "InnoDB: " LOG_ARCHIVE_NAME " is in unrecognized format",
first_lsn); return DB_ERROR;
}
if (!srv_encrypt_log); elseif (!log_crypt_read_header(buf, true)) return DB_ERROR; else
{
buf+= 32/*log_crypt_read_header()*/, n_checkpoint= 4/* 32/8 */; if (!tmp_buf)
{
tmp_buf= static_cast<byte*>
(ut_malloc_dontdump(tmp_buf_size, PSI_INSTRUMENT_ME)); if (!tmp_buf) return DB_OUT_OF_MEMORY;
}
}
}
if (first_lsn != log_sys.get_first_lsn())
{ /* This checkpoint spanned two files, and it therefore should be thelastvalidcheckpointinthefile.Either log_t::archived_switch_recovery()switchedlog_sys.logtopoint tothesecondfile,orlog_t::archived_mmap_switch_recovery_complete()
switched both log_sys.log and log_sys.buf. */
log_sys.archived_switch_recovery_rewind_checkpoint(); break;
}
if (err != DB_SUCCESS) { goto func_exit;
}
} while (missing_tablespace);
rescan = true; /* Because in the loop above we overwrote the initiallystoredrecv_sys.pages,wemust
restart parsing the log from the very beginning. */
/* FIXME: Use a separate loop for checking for tablespaces(notindividualpages),whileretaining
the initial recv_sys.pages. */
mysql_mutex_lock(&recv_sys.mutex);
ut_ad(log_sys.get_flushed_lsn() >= recv_sys.lsn);
recv_sys.clear();
log_sys.recovery_rewind(log_sys.last_checkpoint_lsn);
mysql_mutex_unlock(&recv_sys.mutex);
}
/* The database is now ready to start almost normal processing of user transactions:transactionrollbacksandtheapplicationofthelog
records in the hash table can be run in background. */ if (err == DB_SUCCESS && deferred_spaces.reinit_all()
&& !srv_force_recovery) {
err = DB_CORRUPTION;
}
if (expect_encrypted &&
mach_read_from_4(page + FIL_PAGE_FILE_FLUSH_LSN_OR_KEY_VERSION))
{ if (!fil_space_verify_crypt_checksum(page, space->zip_size())) returnfalse; if (page_type != FIL_PAGE_PAGE_COMPRESSED_ENCRYPTED) returntrue; if (space->zip_size()) returnfalse;
memcpy(tmp_page, page, space->physical_size()); if (!fil_space_decrypt(space, tmp_frame, tmp_page)) returnfalse;
}
switch (page_type) { case FIL_PAGE_PAGE_COMPRESSED:
memcpy(tmp_page, page, space->physical_size()); /* fall through */ case FIL_PAGE_PAGE_COMPRESSED_ENCRYPTED: if (space->zip_size()) returnfalse; /* ROW_FORMAT=COMPRESSED cannot be page_compressed */
ulint decomp= fil_page_decompress(tmp_frame, tmp_page, space->flags); if (!decomp) returnfalse; /* decompression failed */ if (decomp == srv_page_size) returnfalse; /* the page was not compressed (invalid page type) */
page= tmp_page;
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.