/** Table row modification operations during online table rebuild.
Delete-marked records are not copied to the rebuilt table. */ enum row_tab_op { /** Insert a record */
ROW_T_INSERT = 0x41, /** Update a record in place */
ROW_T_UPDATE, /** Delete (purge) a record */
ROW_T_DELETE
};
/** Index record modification operations during online index creation */ enum row_op { /** Insert a record */
ROW_OP_INSERT = 0x61, /** Delete a record */
ROW_OP_DELETE
};
/** Size of the modification log entry header, in bytes */ #define ROW_LOG_HEADER_SIZE 2/*op, extra_size*/
/** Log block for modifications during online ALTER TABLE */ struct row_log_buf_t {
byte* block; /*!< file block buffer */
size_t size; /*!< length of block in bytes */
ut_new_pfx_t block_pfx; /*!< opaque descriptor of "block". Set byut_allocator::allocate_large()andfedto
ut_allocator::deallocate_large(). */
mrec_buf_t buf; /*!< buffer for accessing a record
that spans two blocks */
ulint blocks; /*!< current position in blocks */
ulint bytes; /*!< current position within block */
ulonglong total; /*!< logical position, in bytes from thestartoftherow_log_tablelog; 0forrow_log_online_op()and
row_log_apply(). */
};
/** @brief Buffer for logging modifications during online index creation
Whenhead.blocks==tail.blocks,thereaderwillaccesstail.block directly.Whenalsohead.bytes==tail.bytes,bothcountswillbe
reset to 0 and the file will be truncated. */ struct row_log_t {
pfs_os_file_t fd; /*!< file descriptor */
mysql_mutex_t mutex; /*!< mutex protecting error,
max_trx and tail */
dict_table_t* table; /*!< table that is being rebuilt, orNULLwhenthisisasecondary
index that is being created online */ bool same_pk;/*!< whether the definition of the PRIMARY KEY
has remained the same */ const dtuple_t* defaults; /*!< default values of added, changed columns,
or NULL */ const ulint* col_map;/*!< mapping of old column numbers to
new ones, or NULL if !table */
dberr_t error; /*!< error that occurred during online
table rebuild */ /** The transaction ID of the ALTER TABLE transaction. Any concurrentDMLwouldnecessarilybeloggedwithalarger transactionID,becauseha_innobase::prepare_inplace_alter_table() actsasabarrierthatensuresthatanyconcurrenttransaction thatoperatesonthetablewouldhavebeenstartedafter ha_innobase::prepare_inplace_alter_table()returnsandbefore ha_innobase::commit_inplace_alter_table(commit=true)isinvoked.
Duetothenondeterministicnatureofpurgeandduetothe possibilityofupgradingfromanearlierversionofMariaDB orMySQL,itispossiblethatrow_log_table_low()wouldbe fedDB_TRX_IDthatprecedesthanmin_trx.Wemustnormalize
such references to reset_trx_id[]. */
trx_id_t min_trx;
trx_id_t max_trx;/*!< biggest observed trx_id in row_log_online_op(); protectedbymutexandindex->lockS-latch,
or by index->lock X-latch only */
row_log_buf_t tail; /*!< writer context; protectedbymutexandindex->lockS-latch,
or by index->lock X-latch only */
size_t crypt_tail_size; /*!< size of crypt_tail_size*/
byte* crypt_tail; /*!< writer context; temporarybufferusedinencryption,
decryption or NULL*/
row_log_buf_t head; /*!< reader context; protected by MDL only;
modifiable by row_log_apply_ops() */
size_t crypt_head_size; /*!< size of crypt_tail_size*/
byte* crypt_head; /*!< reader context; temporarybufferusedinencryption,
decryption or NULL */ constchar* path; /*!< where to create temporary file during
log operation */ /** the number of core fields in the clustered index of the sourcetable;beforerow_log_table_apply()completes,the tablecouldbeemptied,sothattable->is_instant()nolongerholds,
but all log records must be in the "instant" format. */ unsigned n_core_fields; /** the default values of non-core fields when the operation started */
dict_col_t::def_t* non_core_fields; bool allow_not_null; /*!< Whether the alter ignore is being usedorifthesqlmodeisnon-strictmode; ifnot,NULLvalueswillnotbeconvertedto
defaults */ const TABLE* old_table; /*< Use old table in case of error. */
uint64_t n_rows; /*< Number of rows read from the table */
/** Alter table transaction. It can be used to apply the DML logs
into the table */ const trx_t* alter_trx;
/** Determine whether the log should be in the 'instant ADD' format @param[in]indextheclusteredindexofthesourcetable
@return whether to use the 'instant ADD COLUMN' format */ bool is_instant(const dict_index_t* index) const
{
ut_ad(table);
ut_ad(n_core_fields <= index->n_fields); return n_core_fields != index->n_fields;
}
/** Create the file or online log if it does not exist. @param[in,out]logonlinerebuildlog
@return true if success, false if not */ static MY_ATTRIBUTE((warn_unused_result))
pfs_os_file_t
row_log_tmpfile(
row_log_t* log)
{
DBUG_ENTER("row_log_tmpfile"); if (log->fd == OS_FILE_CLOSED) {
log->fd = row_merge_file_create_low(log->path);
DBUG_EXECUTE_IF("row_log_tmpfile_fail", if (log->fd != OS_FILE_CLOSED)
row_merge_file_destroy_low(log->fd);
log->fd = OS_FILE_CLOSED;); if (log->fd != OS_FILE_CLOSED) {
MONITOR_ATOMIC_INC(MONITOR_ALTER_TABLE_LOG_FILES);
}
}
DBUG_RETURN(log->fd);
}
/** Allocate the memory for the log buffer. @param[in,out]log_bufBufferusedforlogoperation
@return TRUE if success, false if not */ static MY_ATTRIBUTE((warn_unused_result)) bool
row_log_block_allocate(
row_log_buf_t& log_buf)
{
DBUG_ENTER("row_log_block_allocate"); if (log_buf.block == NULL) {
DBUG_EXECUTE_IF( "simulate_row_log_allocation_failure",
DBUG_RETURN(false);
);
/* Compute the size of the record. This differs from row_merge_buf_encode(),becauseherewedonotencode
extra_size+1 (and reserve 0 as the end-of-chunk marker). */
if (byte_offset + srv_sort_buf_size >= srv_online_max_size) { if (index->online_status != ONLINE_INDEX_COMPLETE) goto write_failed; /* About to run out of log, InnoDB has to
apply the online log for the completed index */
index->lock.s_unlock();
dberr_t error= row_log_apply(
trx, index, nullptr, nullptr);
index->lock.s_lock(SRW_LOCK_CALL); if (error != DB_SUCCESS) { /* Mark all newly added indexes
as corrupted */
log->error = error;
success = false; goto err_exit;
}
/* Recheck whether the index online log */ if (!index->online_log) { goto err_exit;
}
/* If encryption is enabled encrypt buffer before writing it
to file system. */ if (srv_encrypt_log) { if (!log_tmp_block_encrypt(
buf, srv_sort_buf_size,
log->crypt_tail, byte_offset)) {
log->error = DB_DECRYPTION_FAILED; goto write_failed;
}
/******************************************************//**
Gets the error status of the online index rebuild log.
@return DB_SUCCESS or error code */
dberr_t
row_log_table_get_error( /*====================*/ const dict_index_t* index) /*!< in: clustered index of a table
that is being rebuilt online */
{
ut_ad(dict_index_is_clust(index));
ut_ad(dict_index_is_online_ddl(index)); return(index->online_log->error);
}
/******************************************************//**
Starts logging an operation to a table that is being rebuilt.
@return pointer to log, or NULL if no logging is necessary */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
byte*
row_log_table_open( /*===============*/
row_log_t* log, /*!< in/out: online rebuild log */
ulint size, /*!< in: size of log record */
ulint* avail) /*!< out: available size for log record */
{
mysql_mutex_lock(&log->mutex);
if (size > *avail) { /* Make sure log->tail.buf is large enough */
ut_ad(size <= sizeof log->tail.buf); return(log->tail.buf);
} else { return(log->tail.block + log->tail.bytes);
}
}
/******************************************************//**
Stops logging an operation to a table that is being rebuilt. */ static MY_ATTRIBUTE((nonnull)) void
row_log_table_close_func( /*=====================*/
dict_index_t* index, /*!< in/out: online rebuilt index */ #ifdef UNIV_DEBUG const byte* b, /*!< in: end of log record */ #endif/* UNIV_DEBUG */
ulint size, /*!< in: size of log record */
ulint avail) /*!< in: available size for log record */
{
row_log_t* log = index->online_log;
/* If encryption is enabled encrypt buffer before writing it
to file system. */ if (srv_encrypt_log) { if (!log_tmp_block_encrypt(
log->tail.block, srv_sort_buf_size,
log->crypt_tail, byte_offset,
index->table->space_id)) {
log->error = DB_DECRYPTION_FAILED; goto err_exit;
}
/******************************************************//**
Logs a delete operation to a table that is being rebuilt. This will be merged in row_log_table_apply_delete(). */ void
row_log_table_delete( /*=================*/ const rec_t* rec, /*!< in: clustered index leaf page record,
page X-latched */
dict_index_t* index, /*!< in/out: clustered index, S-latched
or X-latched */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec,index) */ const byte* sys) /*!< in: DB_TRX_ID,DB_ROLL_PTR that should
be logged, or NULL to use those in rec */
{
ulint old_pk_extra_size;
ulint old_pk_size;
ulint mrec_size;
ulint avail_size;
mem_heap_t* heap = NULL; const dtuple_t* old_pk;
/* Create the tuple PRIMARY KEY,DB_TRX_ID,DB_ROLL_PTR in new_table. */ if (index->online_log->same_pk) {
dtuple_t* tuple;
ut_ad(new_index->n_uniq == index->n_uniq);
/* The PRIMARY KEY and DB_TRX_ID,DB_ROLL_PTR are in the first
fields of the record. */
heap = mem_heap_create(
DATA_TRX_ID_LEN
+ DTUPLE_EST_ALLOC(new_index->first_user_field()));
old_pk = tuple = dtuple_create(heap,
new_index->first_user_field());
dict_index_copy_types(tuple, new_index, tuple->n_fields);
dtuple_set_n_fields_cmp(tuple, new_index->n_uniq);
for (ulint i = 0; i < dtuple_get_n_fields(tuple); i++) {
ulint len; constvoid* field = rec_get_nth_field(
rec, offsets, i, &len);
dfield_t* dfield = dtuple_get_nth_field(
tuple, i);
ut_ad(len != UNIV_SQL_NULL);
ut_ad(!rec_offs_nth_extern(offsets, i));
dfield_set_data(dfield, field, len);
}
/* old_pk=row_log_table_get_pk() [not needed in INSERT] is a prefix oftheclusteredindexrecord(PRIMARYKEY,DB_TRX_ID,DB_ROLL_PTR),
with no information on virtual columns */
ut_ad(!old_pk || !insert);
ut_ad(!old_pk || old_pk->n_v_fields == 0);
/******************************************************//**
Logs an update to a table that is being rebuilt. This will be merged in row_log_table_apply_update(). */ void
row_log_table_update( /*=================*/ const rec_t* rec, /*!< in: clustered index leaf page record,
page X-latched */
dict_index_t* index, /*!< in/out: clustered index, S-latched
or X-latched */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec,index) */ const dtuple_t* old_pk) /*!< in: row_log_table_get_pk()
before the update */
{
row_log_table_low(rec, index, offsets, false, old_pk);
}
/** Gets the old table column of a PRIMARY KEY column. @paramtableoldtable(beforeALTERTABLE) @paramcol_mapmappingofoldcolumnnumberstonewones @paramcol_nocolumnpositioninthenewtable
@return old table column, or NULL if this is an added column */ static const dict_col_t*
row_log_table_get_pk_old_col( /*=========================*/ const dict_table_t* table, const ulint* col_map,
ulint col_no)
{ for (ulint i = 0; i < table->n_cols; i++) { if (col_no == col_map[i]) { return(dict_table_get_nth_col(table, i));
}
}
return(NULL);
}
/** Maps an old table column of a PRIMARY KEY column. @param[in]ifieldclusteredindexfieldinthenewtable(after ALTERTABLE) @param[in]indextheclusteredindexofifield @param[in,out]dfieldclusteredindextuplefieldinthenewtable @param[in,out]heapmemoryheapforallocatingdfieldcontents @param[in]recclusteredindexleafpagerecordintheold table @param[in]offsetsrec_get_offsets(rec) @param[in]irecfieldcorrespondingtocol @param[in]zip_sizeROW_FORMAT=COMPRESSEDsizeoftheoldtable @param[in]max_lenmaximumlengthofdfield @param[in]logrowlogforthetable @retvalDB_INVALID_NULLifaNULLvalueisencountered
@retval DB_TOO_BIG_INDEX_COL if the maximum prefix length is exceeded */ static
dberr_t
row_log_table_get_pk_col( const dict_field_t* ifield, const dict_index_t* index,
dfield_t* dfield,
mem_heap_t* heap, const rec_t* rec, const rec_offs* offsets,
ulint i,
ulint zip_size,
ulint max_len, const row_log_t* log)
{ const byte* field;
ulint len;
field = rec_get_nth_field(rec, offsets, i, &len);
if (len == UNIV_SQL_DEFAULT) {
field = log->instant_field_value(i, &len);
}
if (len == UNIV_SQL_NULL) { if (!log->allow_not_null) { return(DB_INVALID_NULL);
}
field = static_cast<const byte*>(
log->defaults->fields[col_no].data); if (!field) { return(DB_INVALID_NULL);
}
len = log->defaults->fields[col_no].len;
}
if (rec_offs_nth_extern(offsets, i)) {
ulint field_len = ifield->prefix_len;
byte* blob_field;
if (!field_len) {
field_len = ifield->fixed_len; if (!field_len) {
field_len = max_len + 1;
}
}
/******************************************************//**
Constructs the old PRIMARY KEY and DB_TRX_ID,DB_ROLL_PTR
of a table that is being rebuilt.
@return tuple of PRIMARY KEY,DB_TRX_ID,DB_ROLL_PTR in the rebuilt table, or NULL if the PRIMARY KEY definition does not change */ const dtuple_t*
row_log_table_get_pk( /*=================*/ const rec_t* rec, /*!< in: clustered index leaf page record,
page X-latched */
dict_index_t* index, /*!< in/out: clustered index, S-latched
or X-latched */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec,index) */
byte* sys, /*!< out: DB_TRX_ID,DB_ROLL_PTR for
row_log_table_delete(), or NULL */
mem_heap_t** heap) /*!< in/out: memory heap where allocated */
{
dtuple_t* tuple = NULL;
row_log_t* log = index->online_log;
if (log->same_pk) { /* The PRIMARY KEY columns are unchanged. */ if (sys) { /* Store the DB_TRX_ID,DB_ROLL_PTR. */
ulint trx_id_offs = index->trx_id_offset;
mbminlen = col->mbminlen;
mbmaxlen = col->mbmaxlen;
prtype = col->prtype;
} else { /* No matching column was found in the old table,sothismustbeanaddedcolumn.
Copy the default value. */
ut_ad(log->defaults);
/******************************************************//**
Logs an insert to a table that is being rebuilt. This will be merged in row_log_table_apply_insert(). */ void
row_log_table_insert( /*=================*/ const rec_t* rec, /*!< in: clustered index leaf page record,
page X-latched */
dict_index_t* index, /*!< in/out: clustered index, S-latched
or X-latched */ const rec_offs* offsets)/*!< in: rec_get_offsets(rec,index) */
{
row_log_table_low(rec, index, offsets, true, NULL);
}
/******************************************************//**
Converts a log record to a table row.
@return converted row, or NULL if the conversion fails */ static MY_ATTRIBUTE((nonnull, warn_unused_result)) const dtuple_t*
row_log_table_apply_convert_mrec( /*=============================*/ const mrec_t* mrec, /*!< in: merge record */
dict_index_t* index, /*!< in: index of mrec */ const rec_offs* offsets, /*!< in: offsets of mrec */
row_log_t* log, /*!< in: rebuild context */
mem_heap_t* heap, /*!< in/out: memory heap */
dberr_t* error) /*!< out: DB_SUCCESS or DB_MISSING_HISTORYor
reason of failure */
{
dtuple_t* row;
log->n_rows++;
*error = DB_SUCCESS;
/* This is based on row_build(). */ if (log->defaults) {
row = dtuple_copy(log->defaults, heap); /* dict_table_copy_types() would set the fields to NULL */ for (ulint i = 0; i < dict_table_get_n_cols(log->table); i++) {
dict_col_copy_type(
dict_table_get_nth_col(log->table, i),
dfield_get_type(dtuple_get_nth_field(row, i)));
}
} else {
row = dtuple_create(heap, dict_table_get_n_cols(log->table));
dict_table_copy_types(row, log->table);
}
for (ulint i = 0; i < rec_offs_n_fields(offsets); i++) { const dict_field_t* ind_field
= dict_index_get_nth_field(index, i);
if (ind_field->prefix_len) { /* Column prefixes can only occur in key fields,whichcannotbestoredexternally.For acolumnprefix,thereshouldalsobethefull fieldintheclusteredindextuple.Therow
tuple comprises full fields, not prefixes. */
ut_ad(!rec_offs_nth_extern(offsets, i)); continue;
}
const dict_col_t* col
= dict_field_get_col(ind_field);
if (col->is_dropped()) { /* the column was instantly dropped earlier */
ut_ad(index->table->instant); continue;
}
dfield_set_data(dfield, buf, col->len);
} else { /* field length mismatch should not happen whenrebuildingtheredundantrowformat
table. */
ut_ad(0);
*error = DB_CORRUPTION; return(NULL);
}
}
/* See if any columns were changed to NULL or NOT NULL. */ const dict_col_t* new_col
= dict_table_get_nth_col(log->table, col_no);
ut_ad(new_col->same_format(*col));
switch (error) { case DB_SUCCESS: break; case DB_SUCCESS_LOCKED_REC: /* The row had already been copied to the table. */ return(DB_SUCCESS); default: return(error);
}
ut_ad(dict_index_is_clust(index));
for (n_index += index->type != DICT_CLUSTERED;
(index = dict_table_get_next_index(index)); n_index++) { if (index->type & DICT_FTS) { continue;
}
if (error != DB_SUCCESS) { if (error == DB_DUPLICATE_KEY) {
thr_get_trx(thr)->error_key_num = n_index;
} break;
}
}
return(error);
}
/******************************************************//**
Replays an insert operation on a table that was rebuilt.
@return DB_SUCCESS or error code */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
row_log_table_apply_insert( /*=======================*/
que_thr_t* thr, /*!< in: query graph */ const mrec_t* mrec, /*!< in: record to insert */ const rec_offs* offsets, /*!< in: offsets of mrec */
mem_heap_t* offsets_heap, /*!< in/out: memory heap
that can be emptied */
mem_heap_t* heap, /*!< in/out: memory heap */
row_merge_dup_t* dup) /*!< in/out: for reporting
duplicate key errors */
{
row_log_t*log = dup->index->online_log;
dberr_t error; const dtuple_t* row = row_log_table_apply_convert_mrec(
mrec, dup->index, offsets, log, heap, &error);
switch (error) { case DB_SUCCESS:
ut_ad(row != NULL); break; default:
ut_ad(0); /* fall through */ case DB_INVALID_NULL:
ut_ad(row == NULL); return(error);
}
error = row_log_table_apply_insert_low(
thr, row, offsets_heap, heap, dup); if (error != DB_SUCCESS) { /* Report the erroneous row using the new
version of the table. */
innobase_row_to_mysql(dup->table, log->table, row);
} return(error);
}
/******************************************************//**
Deletes a record from a table that is being rebuilt.
@return DB_SUCCESS or error code */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
row_log_table_apply_delete_low( /*===========================*/
btr_pcur_t* pcur, /*!< in/out: B-tree cursor,
will be trashed */ const rec_offs* offsets, /*!< in: offsets on pcur */
mem_heap_t* heap, /*!< in/out: memory heap */
mtr_t* mtr) /*!< in/out: mini-transaction,
will be committed */
{
dberr_t error;
row_ext_t* ext;
dtuple_t* row;
dict_index_t* index = pcur->index();
if (page_rec_is_infimum(btr_pcur_get_rec(pcur))
|| btr_pcur_get_low_match(pcur) < index->n_uniq) { /* All secondary index entries should be found,becausenew_tableisbeingmodifiedby thisthreadonly,andallindexesshouldbe
updated in sync. */
error = DB_INDEX_CORRUPT; goto err_exit;
}
if (page_rec_is_infimum(btr_pcur_get_rec(&pcur))
|| btr_pcur_get_low_match(&pcur) < index->n_uniq) {
all_done:
mtr_commit(&mtr); /* The record was not found. All done. */ /* This should only happen when an earlier ROW_T_INSERTwasskippedor ROW_T_UPDATEwasinterpretedasROW_T_DELETE
due to BLOBs having been freed by rollback. */ return err;
}
if (memcmp(mrec_trx_id, rec_trx_id,
DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN)) { /* The ROW_T_DELETE was logged for a different PRIMARYKEY,DB_TRX_ID,DB_ROLL_PTR. ThisispossibleifaROW_T_INSERTwasskipped oraROW_T_UPDATEwasinterpretedasROW_T_DELETE becausesomeBLOBsweremissingdueto (1)rollingbacktheinitialinsert,or (2)purgingtheBLOBforalaterROW_T_DELETE (3)purging'oldvalues'foralaterROW_T_UPDATE
or ROW_T_DELETE. */
ut_ad(!log->same_pk); goto all_done;
}
}
if (pk_updated || rec_offs_any_extern(cur_offsets)) { /* If the record contains any externally stored columns,performtheupdatebydeleteandinsert, becausewewillnotwriteanyundologthatwould allowpurgetofreeanyorphanedexternallystored
columns. */
if (pk_updated && log->same_pk) { /* The ROW_T_UPDATE log record should only be writtenwhenthePRIMARYKEYfieldsofthe recorddidnotchangeintheoldtable.We canonlygetachangeofPRIMARYKEYcolumns intherebuilttableifthePRIMARYKEYwas
redefined (!same_pk). */
ut_ad(0);
error = DB_CORRUPTION; goto func_exit;
}
if (dict_table_get_next_index(index)) { /* Construct the row corresponding to the old value of
the record. */
old_row = row_build(
ROW_COPY_DATA, index, btr_pcur_get_rec(&pcur),
cur_offsets, NULL, NULL, NULL, &old_ext, heap);
ut_ad(old_row);
/* Report correct index name for duplicate key error. */ if (error == DB_DUPLICATE_KEY) {
thr_get_trx(thr)->error_key_num = n_index;
}
mtr.start();
index->set_modified(mtr);
}
goto func_exit;
}
/******************************************************//**
Applies an operation to a table that was rebuilt.
@return NULL on failure (mrec corruption) or when out of data;
pointer to next record on success */ static MY_ATTRIBUTE((nonnull, warn_unused_result)) const mrec_t*
row_log_table_apply_op( /*===================*/
que_thr_t* thr, /*!< in: query graph */
ulint new_trx_id_col, /*!< in: position of
DB_TRX_ID in new index */
row_merge_dup_t* dup, /*!< in/out: for reporting
duplicate key errors */
dberr_t* error, /*!< out: DB_SUCCESS
or error code */
mem_heap_t* offsets_heap, /*!< in/out: memory heap
that can be emptied */
mem_heap_t* heap, /*!< in/out: memory heap */ const mrec_t* mrec, /*!< in: merge record */ const mrec_t* mrec_end, /*!< in: end of buffer */
rec_offs* offsets) /*!< in/out: work area
for parsing mrec */
{
row_log_t* log = dup->index->online_log;
dict_index_t* new_index = dict_table_get_first_index(log->table);
ulint extra_size; const mrec_t* next_mrec;
dtuple_t* old_pk;
case ROW_T_DELETE:
extra_size = *mrec++;
ut_ad(mrec < mrec_end);
/* We assume extra_size < 0x100 for the PRIMARY KEY prefix.
For fixed-length PRIMARY key columns, it is 0. */
mrec += extra_size;
/* The ROW_T_DELETE record was converted by
rec_convert_dtuple_to_temp() using new_index. */
ut_ad(!new_index->is_instant());
rec_offs_set_n_fields(offsets, new_index->first_user_field());
rec_init_offsets_temp(mrec, new_index, offsets);
next_mrec = mrec + rec_offs_data_size(offsets); if (next_mrec > mrec_end) { return(NULL);
}
case ROW_T_UPDATE: /* Logically, the log entry consists of the (PRIMARYKEY,DB_TRX_ID)oftheoldvalue(converted tothenewprimarykeydefinition)followedby thenewvalueintheoldtabledefinition.Ifthe definitionofthecolumnsbelongingtoPRIMARYKEY isnotchanged,thelogwillonlycontain
DB_TRX_ID,new_row. */
if (log->same_pk) {
ut_ad(new_index->n_uniq == dup->index->n_uniq);
extra_size = *mrec++;
if (extra_size >= 0x80) { /* Read another byte of extra_size. */
/* Copy the PRIMARY KEY fields from mrec to old_pk. */ for (ulint i = 0; i < new_index->n_uniq; i++) { constvoid* field;
ulint len;
dfield_t* dfield;
ut_ad(!rec_offs_nth_extern(offsets, i));
field = rec_get_nth_field(
mrec, offsets, i, &len);
ut_ad(len != UNIV_SQL_NULL);
dfield = dtuple_get_nth_field(old_pk, i);
dfield_set_data(dfield, field, len);
}
} else { /* We assume extra_size < 0x100
for the PRIMARY KEY prefix. */
mrec += *mrec + 1;
if (mrec > mrec_end) { return(NULL);
}
/* Get offsets for PRIMARY KEY,
DB_TRX_ID, DB_ROLL_PTR. */ /* The old_pk prefix was converted by
rec_convert_dtuple_to_temp() using new_index. */
ut_ad(!new_index->is_instant());
rec_offs_set_n_fields(offsets,
new_index->first_user_field());
rec_init_offsets_temp(mrec, new_index, offsets);
#ifdef HAVE_PSI_STAGE_INTERFACE /** Estimate how much an ALTER TABLE progress should be incremented per oneblockoflogapplied. FortheotherphasesofALTERTABLEweincrementtheprogresswith1per pageprocessed. @returnamountofabstractunitstoaddtowork_completedwhenoneblock oflogisapplied.
*/ inline
ulint
row_log_progress_inc_per_block()
{ /* We must increment the progress once per page (as in srv_page_size,default=innodb_page_size=16KiB).
One block here is srv_sort_buf_size (usually 1MiB). */ const ulint pages_per_block = std::max<ulint>(
ulint(srv_sort_buf_size >> srv_page_size_shift), 1);
/* Multiply by an artificial factor of 6 to even the pace with therestoftheALTERTABLEphases,theyprocesspage_sizeamount
of data faster. */ return(pages_per_block * 6);
}
/** Estimate how much work is to be done by the log apply phase ofanALTERTABLEforthisindex. @param[in]indexindexwhoselogtoassess @returnworktobedonebylog-applyinabstractunits
*/
ulint
row_log_estimate_work( const dict_index_t* index)
{ if (index == NULL || index->online_log == NULL
|| index->online_log_is_dummy()) { return(0);
}
/* This read is not protected by index->online_log->mutex for performancereasons.Wewilleventuallynoticeanyerrorthat
was flagged by a DML thread. */
error = index->online_log->error;
if (error != DB_SUCCESS) { goto func_exit;
}
if (mrec) { /* A partial record was read from the previous block. Copythetemporarybufferfull,aswedonotknowthe lengthoftherecord.Parsesubsequentrecordsfrom thebiggerbufferindex->online_log->head.block
or index->online_log->tail.block. */
memcpy((mrec_t*) mrec_end, next_mrec,
ulint((&index->online_log->head.buf)[1] - mrec_end));
mrec = row_log_table_apply_op(
thr, new_trx_id_col,
dup, &error, offsets_heap, heap,
index->online_log->head.buf,
(&index->online_log->head.buf)[1], offsets); if (error != DB_SUCCESS) { goto func_exit;
} elseif (UNIV_UNLIKELY(mrec == NULL)) { /* The record was not reassembled properly. */ goto corruption;
} /* The record was previously found out to be truncated.Nowthattheparsebufferwasextended,
it should proceed beyond the old end of the buffer. */
ut_a(mrec > mrec_end);
ut_ad(next_mrec <= next_mrec_end); /* The following loop must not be parsing the temporary
buffer, but head.block or tail.block. */
/* mrec!=NULL means that the next record starts from the
middle of the block */
ut_ad((mrec == NULL) == (index->online_log->head.bytes == 0));
#ifdef UNIV_DEBUG if (next_mrec_end - srv_sort_buf_size
== index->online_log->head.block) { /* If tail.bytes == 0, next_mrec_end can also be at
the end of tail.block. */ if (index->online_log->tail.bytes == 0) {
ut_ad(next_mrec == next_mrec_end);
ut_ad(index->online_log->tail.blocks == 0);
ut_ad(index->online_log->head.blocks == 0);
ut_ad(index->online_log->head.bytes == 0);
} else {
ut_ad(next_mrec == index->online_log->head.block
+ index->online_log->head.bytes);
ut_ad(index->online_log->tail.blocks
> index->online_log->head.blocks);
}
} elseif (next_mrec_end - index->online_log->tail.bytes
== index->online_log->tail.block) {
ut_ad(next_mrec == index->online_log->tail.block
+ index->online_log->head.bytes);
ut_ad(index->online_log->tail.blocks == 0);
ut_ad(index->online_log->head.blocks == 0);
ut_ad(index->online_log->head.bytes
<= index->online_log->tail.bytes);
} else {
ut_error;
} #endif/* UNIV_DEBUG */
mrec_end = next_mrec_end;
while (!trx_is_interrupted(trx)) {
mrec = next_mrec;
ut_ad(mrec <= mrec_end);
if (mrec == mrec_end) { /* We are at the end of the log.
Mark the replay all_done. */ if (has_index_lock) { goto all_done;
}
}
if (!has_index_lock) { /* We are applying operations from a different blockthantheonethatisbeingwrittento. Wedonotholdindex->lockinorderto allowotherthreadstoconcurrentlybuffer
modifications. */
ut_ad(mrec >= index->online_log->head.block);
ut_ad(mrec_end == index->online_log->head.block
+ srv_sort_buf_size);
ut_ad(index->online_log->head.bytes
< srv_sort_buf_size);
/* Take the opportunity to do a redo log
checkpoint if needed. */
log_free_check();
} else { /* We are applying operations from the last block. Donotallowotherthreadstobufferanything,
so that we can finally catch up and synchronize. */
ut_ad(index->online_log->head.blocks == 0);
ut_ad(index->online_log->tail.blocks == 0);
ut_ad(mrec_end == index->online_log->tail.block
+ index->online_log->tail.bytes);
ut_ad(mrec >= index->online_log->tail.block);
}
/* This read is not protected by index->online_log->mutex forperformancereasons.Wewilleventuallynoticeany
error that was flagged by a DML thread. */
error = index->online_log->error;
if (error != DB_SUCCESS) { goto func_exit;
} elseif (next_mrec == next_mrec_end) { /* The record happened to end on a block boundary.
Do we have more blocks left? */ if (has_index_lock) { /* The index will be locked while
applying the last block. */ goto all_done;
}
if (clust_index->online_log->n_rows == 0) {
clust_index->online_log->n_rows = new_table->stat_n_rows;
}
clust_index->lock.x_lock(SRW_LOCK_CALL);
if (!clust_index->online_log) {
ut_ad(dict_index_get_online_status(clust_index)
== ONLINE_INDEX_COMPLETE); /* This function should not be called unless rebuildingatableonline.Buildinsomefault
tolerance. */
ut_ad(0);
error = DB_ERROR;
} else {
row_merge_dup_t dup = {
clust_index, trx, table,
clust_index->online_log->col_map, 0
};
/******************************************************//**
Allocate the row log for an index and flag the index for online creation.
@retval trueif success, falseifnot */ bool
row_log_allocate( /*=============*/ const trx_t* trx, /*!< in: the ALTER TABLE transaction */
dict_index_t* index, /*!< in/out: index */
dict_table_t* table, /*!< in/out: new table being rebuilt,
or NULL when creating a secondary index */ bool same_pk,/*!< in: whether the definition of the
PRIMARY KEY has remained the same */ const dtuple_t* defaults, /*!< in: default values of
added, changed columns, or NULL */ const ulint* col_map,/*!< in: mapping of old column
numbers to new ones, or NULL if !table */ constchar* path, /*!< in: where to create temporary file */ const TABLE* old_table, /*!< in: table definition before alter */ constbool allow_not_null) /*!< in: allow null to not-null
conversion */
{
row_log_t* log;
DBUG_ENTER("row_log_allocate");
if (!log->crypt_head || !log->crypt_tail) {
row_log_free(log);
DBUG_RETURN(false);
}
}
index->online_log = log;
if (!table) { /* Assign the clustered index online log to table. ItcanbeusedbyconcurrentDMLtoidentifywhether
the table has any online DDL */
index->table->indexes.start->online_log_make_dummy();
log->alter_trx = trx;
}
/* While we might be holding an exclusive data dictionary lock here,inrow_log_abort_sec()wewillnotalwaysbeholdingit.Use
atomic operations in both cases. */
MONITOR_ATOMIC_INC(MONITOR_ONLINE_CREATE_INDEX);
DBUG_RETURN(true);
}
/******************************************************//**
Free the row log for an index that was being created online. */ void
row_log_free( /*=========*/
row_log_t* log) /*!< in,own: row log */
{
MONITOR_ATOMIC_DEC(MONITOR_ONLINE_CREATE_INDEX);
/* We perform the pessimistic variant of the operations if we alreadyholdindex->lockexclusively.First,searchthe record.Theoperationmayalreadyhavebeenperformed, dependingonwhentherowintheclusteredindexwas
scanned. */
*error = cursor.search_leaf(entry, PAGE_CUR_LE, has_index_lock
? BTR_MODIFY_TREE_ALREADY_LATCHED
: BTR_MODIFY_LEAF, &mtr); if (UNIV_UNLIKELY(*error != DB_SUCCESS)) { goto func_exit;
}
ut_ad(dict_index_get_n_unique(index) > 0); /* This test is somewhat similar to row_ins_must_modify_rec(),
but not identical for unique secondary indexes. */ if (cursor.low_match >= dict_index_get_n_unique(index)
&& !page_rec_is_infimum(btr_cur_get_rec(&cursor))) { /* We have a matching record. */ bool exists = (cursor.low_match
== dict_index_get_n_fields(index)); #ifdef UNIV_DEBUG
rec_t* rec = btr_cur_get_rec(&cursor);
ut_ad(page_rec_is_user_rec(rec));
ut_ad(!rec_get_deleted_flag(rec, page_rec_is_comp(rec))); #endif/* UNIV_DEBUG */
ut_ad(exists || dict_index_is_unique(index));
switch (op) { case ROW_OP_DELETE: if (!exists) { /* The existing record matches the uniquesecondaryindexkey,butthe PRIMARYKEYcolumnsdiffer.So,this exactrecorddoesnotexist.For example,wecoulddetectaduplicate keyerrorinsomeoldindexbefore logginganROW_OP_INSERTforour index.ThisROW_OP_DELETEcouldhave beenloggedforrollingback
TRX_UNDO_INSERT_REC. */ goto func_exit;
}
if (!has_index_lock) { /* This needs a pessimistic operation.
Lock the index tree exclusively. */
mtr_commit(&mtr);
mtr_start(&mtr);
index->set_modified(mtr);
*error = cursor.search_leaf(entry, PAGE_CUR_LE,
BTR_MODIFY_TREE,
&mtr); if (UNIV_UNLIKELY(*error != DB_SUCCESS)) { goto func_exit;
} /* No other thread than the current one isallowedtomodifytheindextree.
Thus, the record should still exist. */
ut_ad(cursor.low_match
>= dict_index_get_n_fields(index));
ut_ad(page_rec_is_user_rec(
btr_cur_get_rec(&cursor)));
}
/* As there are no externally stored fields in asecondaryindexrecord,theparameter
rollback=false will be ignored. */
btr_cur_pessimistic_delete(
error, FALSE, &cursor,
BTR_CREATE_FLAG, false, &mtr); break; case ROW_OP_INSERT: if (exists) { /* The record already exists. There isnothingtobeinserted. Thiscouldhappenwhenprocessing TRX_UNDO_DEL_MARK_RECinstatement rollback:
Theoretically,wecouldalsogeta similarsituationwhenaDELETEoperation
is blocked by a FOREIGN KEY constraint. */ goto func_exit;
}
if (dtuple_contains_null(entry)) { /* The UNIQUE KEY columns match, but thereisaNULLvalueinthekey,and
NULL!=NULL. */ goto insert_the_rec;
}
goto duplicate;
}
} else { switch (op) {
rec_t* rec;
big_rec_t* big_rec; case ROW_OP_DELETE: /* The record does not exist. For example, we coulddetectaduplicatekeyerrorinsomeold indexbeforelogginganROW_OP_INSERTforour index.ThisROW_OP_DELETEcouldbeloggedfor
rolling back TRX_UNDO_INSERT_REC. */ goto func_exit; case ROW_OP_INSERT: if (dict_index_is_unique(index)
&& (cursor.up_match
>= dict_index_get_n_unique(index)
|| cursor.low_match
>= dict_index_get_n_unique(index))
&& (!index->n_nullable
|| !dtuple_contains_null(entry))) {
duplicate: /* Duplicate key */
ut_ad(dict_index_is_unique(index));
row_merge_dup_report(dup, entry->fields);
*error = DB_DUPLICATE_KEY; goto func_exit;
}
insert_the_rec: /* Insert the record. As we are inserting into asecondaryindex,therecannotbeexternally
stored columns (!big_rec). */
*error = btr_cur_optimistic_insert(
BTR_NO_UNDO_LOG_FLAG
| BTR_NO_LOCKING_FLAG
| BTR_CREATE_FLAG,
&cursor, &offsets, &offsets_heap, const_cast<dtuple_t*>(entry),
&rec, &big_rec, 0, NULL, &mtr);
ut_ad(!big_rec); if (*error != DB_FAIL) { break;
}
if (!has_index_lock) { /* This needs a pessimistic operation.
Lock the index tree exclusively. */
mtr_commit(&mtr);
mtr_start(&mtr);
index->set_modified(mtr);
*error = cursor.search_leaf(entry, PAGE_CUR_LE,
BTR_MODIFY_TREE,
&mtr); if (*error != DB_SUCCESS) { break;
}
}
/* We already determined that the recorddidnotexist.Nootherthread thanthecurrentoneisallowedto modifytheindextree.Thus,the
record should still not exist. */
/******************************************************//**
Applies an operation to a secondary index that was being created.
@return NULL on failure (mrec corruption) or when out of data;
pointer to next record on success */ static MY_ATTRIBUTE((nonnull, warn_unused_result)) const mrec_t*
row_log_apply_op( /*=============*/
dict_index_t* index, /*!< in/out: index */
row_merge_dup_t*dup, /*!< in/out: for reporting
duplicate key errors */
dberr_t* error, /*!< out: DB_SUCCESS or error code */
mem_heap_t* offsets_heap, /*!< in/out: memory heap for
allocating offsets; can be emptied */
mem_heap_t* heap, /*!< in/out: memory heap for
allocating data tuples */ bool has_index_lock, /*!< in: true if holding index->lock
in exclusive mode */ const mrec_t* mrec, /*!< in: merge record */ const mrec_t* mrec_end, /*!< in: end of buffer */
rec_offs* offsets) /*!< in/out: work area for
rec_init_offsets_temp() */
if (rec_offs_any_extern(offsets)) { /* There should never be any externally stored fields inasecondaryindex,whichiswhatonlineindex creationisusedfor.Therefore,thelogfilemustbe
corrupted. */ goto corrupted;
}
data_size = rec_offs_data_size(offsets);
mrec += data_size;
if (mrec > mrec_end) { return(NULL);
}
entry = row_rec_to_index_entry_low(
mrec - data_size, index, offsets, heap); /* Online index creation is only implemented for secondary
indexes, which never contain off-page columns. */
ut_ad(dtuple_get_n_ext(entry) == 0);
if (index->is_corrupted()) {
error = DB_INDEX_CORRUPT; goto func_exit;
}
if (UNIV_UNLIKELY(index->online_log->head.blocks
> index->online_log->tail.blocks)) {
unexpected_eof:
ib::error() << "Unexpected end of temporary file for index "
<< index->name;
corruption:
error = DB_CORRUPTION; goto func_exit;
}
if (index->online_log->head.blocks
== index->online_log->tail.blocks) { if (index->online_log->head.blocks) { #ifdef HAVE_FTRUNCATE /* Truncate the file in order to save space. */ if (index->online_log->fd > 0
&& ftruncate(index->online_log->fd, 0) == -1) {
ib::error()
<< "\'" << index->name + 1
<< "\' failed with error "
<< errno << ":" << strerror(errno);
if (mrec) { /* A partial record was read from the previous block. Copythetemporarybufferfull,aswedonotknowthe lengthoftherecord.Parsesubsequentrecordsfrom thebiggerbufferindex->online_log->head.block
or index->online_log->tail.block. */
memcpy((mrec_t*) mrec_end, next_mrec,
ulint((&index->online_log->head.buf)[1] - mrec_end));
mrec = row_log_apply_op(
index, dup, &error, offsets_heap, heap,
has_index_lock, index->online_log->head.buf,
(&index->online_log->head.buf)[1], offsets); if (error != DB_SUCCESS) { goto func_exit;
} elseif (UNIV_UNLIKELY(mrec == NULL)) { /* The record was not reassembled properly. */ goto corruption;
} /* The record was previously found out to be truncated.Nowthattheparsebufferwasextended,
it should proceed beyond the old end of the buffer. */
ut_a(mrec > mrec_end);
ut_ad(next_mrec <= next_mrec_end); /* The following loop must not be parsing the temporary
buffer, but head.block or tail.block. */
/* mrec!=NULL means that the next record starts from the
middle of the block */
ut_ad((mrec == NULL) == (index->online_log->head.bytes == 0));
#ifdef UNIV_DEBUG if (next_mrec_end - srv_sort_buf_size
== index->online_log->head.block) { /* If tail.bytes == 0, next_mrec_end can also be at
the end of tail.block. */ if (index->online_log->tail.bytes == 0) {
ut_ad(next_mrec == next_mrec_end);
ut_ad(index->online_log->tail.blocks == 0);
ut_ad(index->online_log->head.blocks == 0);
ut_ad(index->online_log->head.bytes == 0);
} else {
ut_ad(next_mrec == index->online_log->head.block
+ index->online_log->head.bytes);
ut_ad(index->online_log->tail.blocks
> index->online_log->head.blocks);
}
} elseif (next_mrec_end - index->online_log->tail.bytes
== index->online_log->tail.block) {
ut_ad(next_mrec == index->online_log->tail.block
+ index->online_log->head.bytes);
ut_ad(index->online_log->tail.blocks == 0);
ut_ad(index->online_log->head.blocks == 0);
ut_ad(index->online_log->head.bytes
<= index->online_log->tail.bytes);
} else {
ut_error;
} #endif/* UNIV_DEBUG */
mrec_end = next_mrec_end;
while (!trx_is_interrupted(trx)) {
mrec = next_mrec;
ut_ad(mrec < mrec_end);
if (!has_index_lock) { /* We are applying operations from a different blockthantheonethatisbeingwrittento. Wedonotholdindex->lockinorderto allowotherthreadstoconcurrentlybuffer
modifications. */
ut_ad(mrec >= index->online_log->head.block);
ut_ad(mrec_end == index->online_log->head.block
+ srv_sort_buf_size);
ut_ad(index->online_log->head.bytes
< srv_sort_buf_size);
/* Take the opportunity to do a redo log
checkpoint if needed. */
log_free_check();
} else { /* We are applying operations from the last block. Donotallowotherthreadstobufferanything,
so that we can finally catch up and synchronize. */
ut_ad(index->online_log->head.blocks == 0);
ut_ad(index->online_log->tail.blocks == 0);
ut_ad(mrec_end == index->online_log->tail.block
+ index->online_log->tail.bytes);
ut_ad(mrec >= index->online_log->tail.block);
}
if (error != DB_SUCCESS) { goto func_exit;
} elseif (next_mrec == next_mrec_end) { /* The record happened to end on a block boundary.
Do we have more blocks left? */ if (has_index_lock) { /* The index will be locked while
applying the last block. */ goto all_done;
}
dict_index_set_online_status(index, ONLINE_INDEX_ABORTED);
} elseif (stage) { /* Mark the index as completed only when it is
being called by DDL thread */
ut_ad(dup.n_dup == 0);
dict_index_set_online_status(index, ONLINE_INDEX_COMPLETE);
}
bool success= true;
dict_index_t *index= dict_table_get_next_index(clust_index); while (index)
{
index->lock.s_lock(SRW_LOCK_CALL); if (index->online_log &&
index->online_status <= ONLINE_INDEX_CREATION &&
!index->is_corrupted())
{ if (is_update)
{ /* Ignore the index if the update doesn't affect the index */ if (!row_upd_changes_ord_field_binary(index, update,
nullptr,
row, new_ext)) goto next_index;
dtuple_t *old_entry= row_build_index_entry_low(
old_row, old_ext, index, heap, ROW_BUILD_NORMAL);
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.0.108Bemerkung:
(vorverarbeitet am 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.