/**************************************************//**
@file row/row0merge.cc New index creation routines using a merge sort
Created 12/4/2005 Jan Lindstrom
Completed by Sunny Bains and Marko Makela
*******************************************************/ #include <my_global.h> #include <log.h> #include <sql_class.h> #include <math.h>
/* Ignore posix_fadvise() on those platforms where it does not exist */ #ifdefined _WIN32 # define posix_fadvise(fd, offset, len, advice) /* nothing */ #endif/* _WIN32 */
/* Whether to disable file system cache */ char srv_disable_sort_file_cache;
/** Class that caches spatial index row tuples made from a single cluster
index page scan, and then insert into corresponding index tree */ class spatial_index_info { public: /** constructor
@param index spatial index to be created */
spatial_index_info(dict_index_t *index) : index(index)
{
ut_ad(index->is_spatial());
}
/** Caches an index row into index tuple vector @param[in]rowtablerow
@param[in] ext externally stored column prefixes, or NULL */ void add(const dtuple_t *row, const row_ext_t *ext, mem_heap_t *heap)
{
dtuple_t *dtuple= row_build_index_entry(row, ext, index, heap);
ut_ad(dtuple);
ut_ad(dtuple->n_fields == index->n_fields); if (ext)
{ /* Replace any references to ext, because ext will be allocated
from row_heap. */ for (ulint i= 1; i < dtuple->n_fields; i++)
{
dfield_t &dfield= dtuple->fields[i]; if (dfield.data >= ext->buf &&
dfield.data <= &ext->buf[ext->n_ext * ext->max_len])
dfield_dup(&dfield, heap);
}
}
m_dtuple_vec.push_back(dtuple);
}
/** Insert spatial index rows cached in vector into spatial index @param[in]trx_idtransactionid @param[in]pcurclusterindexscanningcursor @param[in,out]mtr_startedwhetherscan_mtrisactive @param[in,out]heaptemporarymemoryheap @param[in,out]scan_mtrmini-transactionforpcur
@return DB_SUCCESS if successful, else error number */
dberr_t insert(trx_id_t trx_id, btr_pcur_t* pcur, bool& mtr_started, mem_heap_t* heap, mtr_t* scan_mtr)
{
big_rec_t* big_rec;
rec_t* rec;
btr_cur_t ins_cur;
mtr_t mtr{scan_mtr->trx};
rtr_info_t rtr_info;
rec_offs* ins_offsets = NULL;
dberr_t error = DB_SUCCESS;
dtuple_t* dtuple; const ulint flag = BTR_NO_UNDO_LOG_FLAG
| BTR_NO_LOCKING_FLAG
| BTR_KEEP_SYS_FLAG | BTR_CREATE_FLAG;
private: /** Cache index rows made from a cluster index scan. Usually
for rows on single cluster index page */ typedef std::vector<dtuple_t*, ut_allocator<dtuple_t*> > idx_tuple_vec;
/** vector used to cache index rows made from cluster index scan */
idx_tuple_vec m_dtuple_vec; public: /** the index being built */
dict_index_t*const index;
};
/* Maximum pending doc memory limit in bytes for a fts tokenization thread */ #define FTS_PENDING_DOC_MEMORY_LIMIT 1000000
/** Encode an index record.
@return size of the record */ static MY_ATTRIBUTE((nonnull))
ulint
row_merge_buf_encode( /*=================*/
byte** b, /*!< in/out: pointer to
current end of output buffer */ const dict_index_t* index, /*!< in: index */ const mtuple_t* entry, /*!< in: index fields
of the record to encode */
ulint n_fields) /*!< in: number of fields
in the entry */
{
ulint size;
ulint extra_size;
for (ulint i = 0; i < n_fields; i++, field++, ifield++)
{
dfield_copy(field, &row.fields[i]);
ulint len= dfield_get_len(field); const dict_col_t* const col= ifield->col;
if (dfield_is_null(field)) continue;
ulint fixed_len= ifield->fixed_len;
/* CHAR in ROW_FORMAT=REDUNDANT is always fixed-length,butinthetemporaryfileitis
variable-length for variable-length character sets. */ if (fixed_len && !index->table->not_redundant() &&
col->mbminlen != col->mbmaxlen)
fixed_len= 0;
/* Add to the total size of the record in row_merge_block_t theencodedlengthofextra_sizeandtheextrabytes(extra_size). Seerow_merge_buf_write()forthevariable-lengthencoding
of extra_size. */
data_size += (extra_size + 1) + ((extra_size + 1) >= 0x80);
/* Reserve bytes for the end marker of row_merge_block_t. */ if (buf->total_size + data_size >= srv_sort_buf_size) return0;
buf->total_size += data_size;
buf->n_tuples++;
field= entry->fields;
do
dfield_dup(field++, buf->heap); while (--n_fields);
DBUG_EXECUTE_IF( "ib_row_merge_buf_add_two", if (buf->n_tuples >= 2) DBUG_RETURN(0););
UNIV_PREFETCH_R(row->fields);
/* If we are building FTS index, buf->index points to the'fts_sort_idx',andrealFTSindexisstoredin
fts_index */
index = (buf->index->type & DICT_FTS) ? fts_index : buf->index;
/* create spatial index should not come here */
ut_ad(!dict_index_is_spatial(index));
for (i = 0; i < n_fields; i++, field++, ifield++) {
ulint len;
ulint fixed_len; const dfield_t* row_field; const dict_col_t* const col = ifield->col; const dict_v_col_t* const v_col = col->is_virtual()
? reinterpret_cast<const dict_v_col_t*>(col)
: NULL;
/* Process the Doc ID column */ if (!v_col && *doc_id
&& col->ind == index->table->fts->doc_col) {
fts_write_doc_id((byte*) &write_doc_id, *doc_id);
/* Note: field->data now points to a value on the stack:&write_doc_idafterdfield_set_data().Because thereisonlyonedoc_idperrow,itshouldn'tmatter. Weallocateanewbufferbeforeweleavethefunction
later below. */
/* Copy the column collation to the
tuple field */ if (col_collate) { auto it = col_collate->find(col->ind); if (it != col_collate->end()) {
field->type
.assign(*it->second);
}
}
}
/* Tokenize and process data for FTS */ if (index->type & DICT_FTS) {
fts_doc_item_t* doc_item;
byte* value; void* ptr; const ulint max_trial_count = 10000;
ulint trial_count = 0;
/* fetch Doc ID if it already exists intherow,andnotsuppliedbythe caller.Evenifthevaluecolumnis NULL,westillneedtogettheDoc IDsotomaintainthecorrectmax
Doc ID */ if (*doc_id == 0) { const dfield_t* doc_field;
doc_field = dtuple_get_nth_field(
row,
index->table->fts->doc_col);
*doc_id = (doc_id_t) mach_read_from_8( static_cast<const byte*>(
dfield_get_data(doc_field)));
if (*doc_id == 0) {
ib::warn() << "FTS Doc ID is" " zero. Record" " skipped"; goto error;
}
}
if (dfield_is_null(field)) {
n_row_added = 1; continue;
}
fixed_len = ifield->fixed_len; if (fixed_len && !dict_table_is_comp(index->table)
&& col->mbminlen != col->mbmaxlen) { /* CHAR in ROW_FORMAT=REDUNDANT is always fixed-length,butinthetemporaryfileitis variable-lengthforvariable-lengthcharacter
sets. */
fixed_len = 0;
}
if (fixed_len) { #ifdef UNIV_DEBUG /* len should be between size calculated based on
mbmaxlen and mbminlen */
ut_ad(len <= fixed_len);
ut_ad(!col->mbmaxlen || len >= col->mbminlen
* (fixed_len / col->mbmaxlen));
ut_ad(!dfield_is_ext(field)); #endif/* UNIV_DEBUG */
} elseif (dfield_is_ext(field)) {
extra_size += 2;
} elseif (len < 128
|| (!DATA_BIG_COL(col))) {
extra_size++;
} else { /* For variable-length columns, we look up the maximumlengthfromthecolumnitself.Ifthis isaprefixindexcolumnshorterthan256bytes,
this will waste one byte. */
extra_size += 2;
}
data_size += len;
}
/* If this is FTS index, we already populated the sort buffer, return
here */ if (index->type & DICT_FTS) { goto end;
}
/* Add to the total size of the record in row_merge_block_t theencodedlengthofextra_sizeandtheextrabytes(extra_size). Seerow_merge_buf_write()forthevariable-lengthencoding
of extra_size. */
data_size += (extra_size + 1) + ((extra_size + 1) >= 0x80);
/* Record size can exceed page size while converting to redundantrowformat.Butthereisassert ut_ad(size<srv_page_size)inrec_offs_data_size().
It may hit the assert before attempting to insert the row. */ if (conv_heap != NULL && data_size > srv_page_size) {
*err = DB_TOO_BIG_RECORD;
}
ut_ad(data_size < srv_sort_buf_size);
/* Reserve bytes for the end marker of row_merge_block_t. */ if (buf->total_size + data_size >= srv_sort_buf_size) { goto error;
}
do {
dfield_dup(field++, buf->heap);
} while (--n_fields);
if (conv_heap != NULL) {
mem_heap_empty(conv_heap);
}
end: if (vcol_storage.innobase_record)
innobase_free_row_for_vcol(&vcol_storage);
DBUG_RETURN(n_row_added);
}
/*************************************************************//**
Report a duplicate key. */ void
row_merge_dup_report( /*=================*/
row_merge_dup_t* dup, /*!< in/out: for reporting duplicates */ const dfield_t* entry) /*!< in: duplicate index entry */
{ if (!dup->n_dup++ && dup->table) { /* Only report the first duplicate record,
but count all duplicate records. */
innobase_fields_to_mysql(dup->table, dup->index, entry);
}
}
/*************************************************************//**
Compare two tuples.
@return positive, 0, negative if a is greater, equal, less, than b,
respectively */ static MY_ATTRIBUTE((warn_unused_result)) int
row_merge_tuple_cmp( /*================*/ const dict_index_t* index, /*< in: index tree */
ulint n_uniq, /*!< in: number of unique fields */
ulint n_field,/*!< in: number of fields */ const mtuple_t& a, /*!< in: first tuple to be compared */ const mtuple_t& b, /*!< in: second tuple to be compared */
row_merge_dup_t* dup) /*!< in/out: for reporting duplicates,
NULL if non-unique index */
{ int cmp; const dfield_t* af = a.fields; const dfield_t* bf = b.fields;
ulint n = n_uniq; const dict_field_t* f = index->fields;
ut_ad(n_uniq > 0);
ut_ad(n_uniq <= n_field);
/* Compare the fields of the tuples until a difference is foundorwerunoutoffieldstocompare.If!cmpatthe
end, the tuples are equal. */ do {
cmp = cmp_dfield_dfield(af++, bf++, (f++)->descending);
} while (!cmp && --n);
if (cmp) { return(cmp);
}
if (dup) { /* Report a duplicate value error if the tuples are logicallyequal.NULLcolumnsarelogicallyinequal, althoughtheyareequalinthesortingorder.Find
out if any of the fields are NULL. */ for (const dfield_t* df = a.fields; df != af; df++) { if (dfield_is_null(df)) { goto no_report;
}
}
row_merge_dup_report(dup, a.fields);
}
no_report: /* The n_uniq fields were equal, but we compare all fields so
that we will get the same (internal) order as in the B-tree. */ for (n = n_field - n_uniq + 1; --n; ) {
cmp = cmp_dfield_dfield(af++, bf++, (f++)->descending); if (cmp) { return(cmp);
}
}
/* This should never be reached, except in a secondary index whencreatingasecondaryindexandaPRIMARYKEY,andthere isaduplicateinthePRIMARYKEYthathasnotbeendetected
yet. Internally, an index must never contain duplicates. */ return(cmp);
}
/** Wrapper for row_merge_tuple_sort() to inject some more context to UT_SORT_FUNCTION_BODY(). @paramtuplesarrayoftuplesthatbeingsorted @paramauxworkarea,samesizeastuples[] @paramlowlowerboundofthesortingarea,inclusive
@param high upper bound of the sorting area, inclusive */ #define row_merge_tuple_sort_ctx(tuples, aux, low, high) \
row_merge_tuple_sort(index,n_uniq,n_field,dup, tuples, aux, low, high) /** Wrapper for row_merge_tuple_cmp() to inject some more context to UT_SORT_FUNCTION_BODY(). @paramafirsttupletobecompared @parambsecondtupletobecompared @returnpositive,0,negative,ifaisgreater,equal,less,thanb,
respectively */ #define row_merge_tuple_cmp_ctx(a,b) \
row_merge_tuple_cmp(index, n_uniq, n_field, a, b, dup)
/**********************************************************************//**
Merge sort the tuple buffer in main memory. */ static void
row_merge_tuple_sort( /*=================*/ const dict_index_t* index, /*!< in: index tree */
ulint n_uniq, /*!< in: number of unique fields */
ulint n_field,/*!< in: number of fields */
row_merge_dup_t* dup, /*!< in/out: reporter of duplicates
(NULL if non-unique index) */
mtuple_t* tuples, /*!< in/out: tuples */
mtuple_t* aux, /*!< in/out: work area */
ulint low, /*!< in: lower bound of the
sorting area, inclusive */
ulint high) /*!< in: upper bound of the
sorting area, exclusive */
{
ut_ad(n_field > 0);
ut_ad(n_uniq <= n_field);
/* For encrypted tables, encrypt data before writing */ if (srv_encrypt_log) { if (!log_tmp_block_encrypt(static_cast<const byte*>(buf),
buf_len, static_cast<byte*>(crypt_buf),
ofs)) {
DBUG_RETURN(false);
}
#ifdef POSIX_FADV_DONTNEED /* The block will be needed on the next merge pass,
but it can be evicted from the file cache meanwhile. */
posix_fadvise(fd, ofs, buf_len, POSIX_FADV_DONTNEED); #endif/* POSIX_FADV_DONTNEED */
DBUG_RETURN(success);
}
/********************************************************************//**
Read a merge record.
@return pointer to next record, or NULL on I/O error or end of list */ const byte*
row_merge_read_rec( /*===============*/
row_merge_block_t* block, /*!< in/out: file buffer */
mrec_buf_t* buf, /*!< in/out: secondary buffer */ const byte* b, /*!< in: pointer to record */ const dict_index_t* index, /*!< in: index of the record */ const pfs_os_file_t& fd, /*!< in: file descriptor */
ulint* foffs, /*!< in/out: file offset */ const mrec_t** mrec, /*!< out: pointer to merge record, orNULLonendoflist
(non-NULL on I/O error) */
rec_offs* offsets,/*!< out: offsets of mrec */
row_merge_block_t* crypt_block, /*!< in: crypt buf or NULL */
ulint space) /*!< in: space id */
{
ulint extra_size;
ulint data_size;
ulint avail_size;
/* Normalize extra_size. Above, value 0 signals "end of list". */
extra_size--;
/* Read the extra bytes. */
if (UNIV_UNLIKELY(b + extra_size >= &block[srv_sort_buf_size])) { /* The record spans two blocks. Copy the entire record totheauxiliarybufferandhandlethisasaspecial
case. */
if (!row_merge_read(fd, ++(*foffs), block,
crypt_block,
space)) {
goto err_exit;
}
/* Wrap around to the beginning of the buffer. */
b = &block[0];
/* Copy the record. */
memcpy(*buf + avail_size, b, extra_size - avail_size);
b += extra_size - avail_size;
*mrec = *buf + extra_size;
rec_init_offsets_temp(*mrec, index, offsets);
data_size = rec_offs_data_size(offsets);
/* These overflows should be impossible given that recordsaremuchsmallerthaneitherbuffer,and
the record starts near the beginning of each buffer. */
ut_a(extra_size + data_size < sizeof *buf);
ut_a(b + data_size < &block[srv_sort_buf_size]);
/* Copy the data bytes. */
memcpy(*buf + extra_size, b, data_size);
b += data_size;
if (UNIV_UNLIKELY(b + size >= &block[srv_sort_buf_size])) { /* The record spans two blocks.
Copy it to the temporary buffer first. */
avail_size = ulint(&block[srv_sort_buf_size] - b);
/* Copy the head of the temporary buffer, write thecompletedblock,andcopythetailofthe
record to the head of the new block. */
memcpy(b, buf[0], avail_size);
if (!row_merge_write(fd, (*foffs)++, block,
crypt_block,
space)) { return(NULL);
}
MEM_UNDEFINED(&block[0], srv_sort_buf_size);
/* Copy the rest. */
b = &block[0];
memcpy(b, buf[0] + avail_size, size - avail_size);
b += size - avail_size;
} else {
row_merge_write_rec_low(b, extra_size, size, fd, *foffs,
mrec, offsets);
b += size;
}
return(b);
}
/********************************************************************//**
Write an end-of-list marker.
@return pointer to end of block, or NULL on error */ static
byte*
row_merge_write_eof( /*================*/
row_merge_block_t* block, /*!< in/out: file buffer */
byte* b, /*!< in: pointer to end of block */ const pfs_os_file_t& fd, /*!< in: file descriptor */
ulint* foffs, /*!< in/out: file offset */
row_merge_block_t* crypt_block, /*!< in: crypt buf or NULL */
ulint space) /*!< in: space id */
{
ut_ad(block);
ut_ad(b >= &block[0]);
ut_ad(b < &block[srv_sort_buf_size]);
ut_ad(foffs);
/** Create a temporary file if it has not been created already. @param[in,out]tmpfdtemporaryfilehandle @param[in]pathlocationforcreatingtemporaryfile
@return true on success, false on error */ static MY_ATTRIBUTE((warn_unused_result)) bool
row_merge_tmpfile_if_needed(
pfs_os_file_t* tmpfd, constchar* path)
{ if (*tmpfd == OS_FILE_CLOSED) {
*tmpfd = row_merge_file_create_low(path); if (*tmpfd != OS_FILE_CLOSED) {
MONITOR_ATOMIC_INC(MONITOR_ALTER_TABLE_SORT_FILES);
}
}
return(*tmpfd != OS_FILE_CLOSED);
}
/** Create a temporary file for merge sort if it was not created already. @param[in,out]filemergefilestructure @param[in]nrecnumberofrecordsinthefile @param[in]pathlocationforcreatingtemporaryfile
@return true on success, false on error */ static MY_ATTRIBUTE((warn_unused_result)) bool
row_merge_file_create_if_needed(
merge_file_t* file,
pfs_os_file_t* tmpfd,
ulint nrec, constchar* path)
{
ut_ad(file->fd == OS_FILE_CLOSED || *tmpfd != OS_FILE_CLOSED); if (file->fd == OS_FILE_CLOSED && row_merge_file_create(file, path)!= OS_FILE_CLOSED) {
MONITOR_ATOMIC_INC(MONITOR_ALTER_TABLE_SORT_FILES); if (!row_merge_tmpfile_if_needed(tmpfd, path) ) { return(false);
}
/* If Doc ID does not exist in the table itself,
fetch the first FTS Doc ID */ if (add_doc_id) {
err = fts_get_next_doc_id(
(dict_table_t*) new_table,
&doc_id, trx->mysql_thd);
ut_ad(err || doc_id > 0);
}
for (ulint i = 0; i < n_index; i++) { if (dict_index_is_spatial(index[i])) {
sp_tuples[count]
= UT_NEW_NOKEY(
spatial_index_info(index[i]));
count++;
}
}
ut_ad(count == num_spatial);
}
mtr.start();
mtr_started = true;
/* Find the clustered index and create a persistent cursor
based on that. */
err = pcur.open_leaf(true, clust_index, BTR_SEARCH_LEAF, &mtr); if (err != DB_SUCCESS) {
err_exit:
trx->error_key_num = 0; goto func_exit;
} else { const page_t* const page = btr_pcur_get_page(&pcur); constauto comp = page_is_comp(page); const rec_t* const rec = comp
? page_rec_next_get<true>(page,
btr_pcur_get_rec(&pcur))
: page_rec_next_get<false>(page,
btr_pcur_get_rec(&pcur)); if (!rec) {
corrupted_metadata:
err = DB_CORRUPTION; goto err_exit;
} if (rec_get_info_bits(rec, comp) & REC_INFO_MIN_REC_FLAG) { if (!clust_index->is_instant()) { goto corrupted_metadata;
} if (comp
&& rec_get_status(rec) != REC_STATUS_INSTANT) { goto corrupted_metadata;
} /* Skip the metadata pseudo-record. */
btr_pcur_get_page_cur(&pcur)->rec = const_cast<rec_t*>(rec);
} elseif (clust_index->is_instant()) { goto corrupted_metadata;
}
}
/* Check if the table is supposed to be empty for our read view.
Weareholdingaclusteredindexleafpagelatchhere. ThatwillobviouslypreventanyconcurrentINSERTfrom
updating bulk_trx_id while we read it. */ if (!online) {
} elseif (trx_id_t bulk_trx_id = old_table->bulk_trx_id) {
ut_ad(trx->read_view.is_open());
ut_ad(bulk_trx_id != trx->id); if (!trx->read_view.changes_visible(bulk_trx_id)) { goto func_exit;
}
}
if (old_table != new_table) { /* The table is being rebuilt. Identify the columns thatwereflaggedNOTNULLinthenewtable,sothat wecanquicklycheckthattherecordsintheoldtable
do not violate the added NOT NULL constraints. */
/* Scan the clustered index. */ for (;;) { /* Do not continue if table pages are still encrypted */ if (!old_table->is_readable() || !new_table->is_readable()) {
err = DB_DECRYPTION_FAILED; goto err_exit;
}
/* The old version must necessarily be inthe"prehistory",becausethe exclusivelockin ha_innobase::prepare_inplace_alter_table() forcedthecompletionofanytransactions
that accessed this table. */
ut_ad(row_get_rec_trx_id(old_vers, clust_index,
offsets) < trx->id);
rec = old_vers;
rec_trx_id = 0;
}
if (rec_get_deleted_flag(
rec,
dict_table_is_comp(old_table))) { /* In delete-marked records, DB_TRX_ID must alwaysrefertoanexistingundologrecord. Above,wedidresetrec_trx_id=0
for rec = old_vers.*/
ut_ad(rec == page_cur_get_rec(cur)
? rec_trx_id
: !rec_trx_id); /* This record was deleted in the latest committedversion,oritwasdeletedand thenreinserted-by-updatebeforepurge
kicked in. Skip it. */ continue;
}
ut_ad(!rec_offs_any_null_extern(rec, offsets));
} elseif (rec_get_deleted_flag(
rec, dict_table_is_comp(old_table))) { /* In delete-marked records, DB_TRX_ID must
always refer to an existing undo log record. */
ut_d(rec_trx_id = rec_get_trx_id(rec, clust_index));
ut_ad(rec_trx_id); /* This must be a purgeable delete-marked record, andthetransactionthatdelete-markedtherecord musthavebeencommittedbeforethis
!online ALTER TABLE transaction. */
ut_ad(rec_trx_id < trx->id); /* Skip delete-marked records.
Skippingdelete-markedrecordswillmakethe createdindexesunuseablefortransactions whosereadviewswerecreatedbeforetheindex creationcompleted,butanattempttopreserve thehistorywouldmakeittrickytodetect
duplicate keys. */ continue;
} else {
offsets = rec_get_offsets(rec, clust_index, NULL,
clust_index->n_core_fields,
ULINT_UNDEFINED, &row_heap); /* This is a locking ALTER TABLE.
Ifwearerebuildingthetable,the DB_TRX_ID,DB_ROLL_PTRshouldbereset,because
there will be no history available. */
ut_ad(rec_get_trx_id(rec, clust_index) < trx->id);
rec_trx_id = 0;
}
/* When !online, we are holding a lock on old_table, preventing anyinsertsthatcouldhavewrittenarecord'stub'before
writing out off-page columns. */
ut_ad(!rec_offs_any_null_extern(rec, offsets));
for (ulint k = 0, i = 0; i < n_index; i++, skip_sort = false) {
row_merge_buf_t* buf = merge_buf[i];
ulint rows_added = 0;
if (dict_index_is_spatial(buf->index)) { if (!row) { continue;
}
ut_ad(sp_tuples[s_idx_cnt]->index
== buf->index);
/* If the geometry field is invalid, report
error. */ if (!row_geo_field_is_valid(row, buf->index)) {
err = DB_CANT_CREATE_GEOMETRY_OBJECT;
trx->error_key_num = i; break;
}
ut_ad(i == 0);
ut_ad(dict_index_is_clust(merge_buf[0]->index)); /* Detect duplicates by comparing the currentrecordwithpreviousrecord. Whentempfileisnotused,records
should be in sorted order. */ if (prev_mtuple.fields != NULL
&& (row_mtuple_cmp(
&prev_mtuple, curr,
&clust_dup) == 0)) {
if (buf->index->type & DICT_FTS) { if (!row || !doc_id) { continue;
}
}
/* The buffer must be sufficiently large toholdatleastonerecord.Itmayonly beemptywhenwereachtheendofthe clusteredindex.row_merge_buf_add()
must not have been called in this loop. */
ut_ad(buf->n_tuples || row == NULL);
/* We have enough data tuples to form a block. Sortthemandwritetodiskiftempfileisused
or insert into index if temp file is not used. */
ut_ad(old_table == new_table
? !dict_index_is_clust(buf->index)
: (i == 0) == dict_index_is_clust(buf->index));
/* We have enough data tuples to form a block.
Sort them (if !skip_sort) and write to disk. */
if (buf->n_tuples) { if (skip_sort) { /* Temporary File is not used.
so insert sorted block to the index */ if (row != NULL) { /* We have to do insert the cachedspatialindexrows,since afterthemtr_commit,thecluster indexpagecouldbeupdated,then thedataincachedrowsbecome
invalid. */
err = row_merge_spatial_rows(
trx->id, sp_tuples,
num_spatial,
row_heap,
&pcur, mtr_started,
&mtr);
if (err != DB_SUCCESS) { goto func_exit;
}
/* We are not at the end of thescanyet.Wemust mtr.commit()inordertobe abletocalllog_free_check() inrow_merge_insert_index_tuples(). Duetomtr.commit(),the currentrowwillbeinvalid,and wemustrereaditonthenext
loop iteration. */ if (mtr_started) { if (!btr_pcur_move_to_prev_on_page(&pcur)) {
err = DB_CORRUPTION; goto func_exit;
}
btr_pcur_store_position(
&pcur, &mtr);
if (max_trx_id > index->trx_id) {
index->trx_id = max_trx_id;
}
index->lock.x_unlock();
}
/* Secondary index and clustered index which is notinsortedordercanusethetemporaryfile.
Fulltext index should not use the temporary file. */ if (!skip_sort && !(buf->index->type & DICT_FTS)) { /* In case we can have all rows in sort buffer, wecaninsertdirectlyintotheindexwithout temporaryfileifclusteredindexdoesnotuses
temporary file. */ if (row == NULL && file->fd == OS_FILE_CLOSED
&& !clust_temp_file) {
DBUG_EXECUTE_IF( "row_merge_write_failure",
err = DB_TEMP_FILE_WRITE_FAIL;
trx->error_key_num = i; goto all_done;);
/* Ensure that duplicates in the clusteredindexwillbedetectedbefore
inserting secondary index records. */ if (dict_index_is_clust(buf->index)) {
clust_temp_file = true;
}
if (prev_fields) {
ut_free(prev_fields);
mem_heap_free(mtuple_heap);
}
if (v_heap) {
mem_heap_free(v_heap);
}
if (conv_heap != NULL) {
mem_heap_free(conv_heap);
}
if (UNIV_LIKELY_NULL(fts_parallel_sort_cond)) {
wait_again: /* Check if error occurs in child thread */ for (ulint j = 0; j < fts_sort_pll_degree; j++) { if (psort_info[j].error != DB_SUCCESS) {
err = psort_info[j].error;
trx->error_key_num = j; break;
}
}
/* Tell all children that parent has done scanning */ for (ulint i = 0; i < fts_sort_pll_degree; i++) { if (err == DB_SUCCESS) {
psort_info[i].state = FTS_PARENT_COMPLETE;
} else {
psort_info[i].state = FTS_PARENT_EXITING;
}
}
/* Now wait all children to report back to be completed */
timespec abstime;
set_timespec(abstime, 1);
mysql_mutex_lock(&psort_info[0].mutex);
my_cond_timedwait(fts_parallel_sort_cond,
&psort_info[0].mutex.m_mutex, &abstime);
mysql_mutex_unlock(&psort_info[0].mutex);
for (ulint i = 0; i < fts_sort_pll_degree; i++) { if (!psort_info[i].child_status) { goto wait_again;
}
}
for (ulint i = 0; i < n_index; i++) {
row_merge_buf_free(merge_buf[i]);
}
row_fts_free_pll_merge_buf(psort_info);
ut_free(merge_buf);
ut_free(pcur.old_rec_buf);
if (sp_tuples != NULL) { for (ulint i = 0; i < num_spatial; i++) {
UT_DELETE(sp_tuples[i]);
}
ut_free(sp_tuples);
}
/* Update the next Doc ID we used. Table should be locked, so
no concurrent DML */ if (max_doc_id && err == DB_SUCCESS) { /* Sync fts cache for other fts indexes to keep all
fts indexes consistent in sync_doc_id. */
err = fts_sync_table(const_cast<dict_table_t*>(new_table), true, trx->mysql_thd);
if (err == DB_SUCCESS) {
new_table->fts->cache->synced_doc_id = max_doc_id;
/* Update the max value as next FTS_DOC_ID */ if (max_doc_id >= new_table->fts->cache->next_doc_id) {
new_table->fts->cache->next_doc_id =
max_doc_id + 1;
}
if (crypt_block) {
MEM_CHECK_ADDRESSABLE(&crypt_block[0], 3 * srv_sort_buf_size);
}
ut_ad(ihalf < file->offset);
of.fd = *tmpfd;
of.offset = 0;
of.n_rec = 0;
#ifdef POSIX_FADV_SEQUENTIAL /* The input file will be read sequentially, starting from the beginningandthemiddle.InLinux,thePOSIX_FADV_SEQUENTIAL
affects the entire file. Each block will be read exactly once. */
posix_fadvise(file->fd, 0, 0,
POSIX_FADV_SEQUENTIAL | POSIX_FADV_NOREUSE); #endif/* POSIX_FADV_SEQUENTIAL */
/* Merge blocks to the output file. */
foffs0 = 0;
foffs1 = ihalf;
if (UNIV_UNLIKELY(of.n_rec != file->n_rec)) { return(DB_CORRUPTION);
}
ut_ad(n_run <= *num_run);
*num_run = n_run;
/* Each run can contain one or more offsets. As merge goes on, thenumberofruns(tomerge)willreduceuntilwehaveone singlerun.Sothenumberofrunswillalwaysbesmallerthan
the number of offsets in file */
ut_ad((*num_run) <= file->offset);
/* The number of offsets in output file is always equal or
smaller than input file */
ut_ad(of.offset <= file->offset);
/* Swap file descriptors for the next pass. */
*tmpfd = file->fd;
*file = of;
MEM_UNDEFINED(&block[0], 3 * srv_sort_buf_size);
return(DB_SUCCESS);
}
/** Merge disk files. @param[in]trxtransaction @param[in]dupdescriptorofindexbeingcreated @param[in,out]filefilecontainingindexentries @param[in,out]block3buffers @param[in,out]tmpfdtemporaryfilehandle @param[in,out]stageperformanceschemaaccountingobject,usedby ALTERTABLE.IfnotNULL,stage->begin_phase_sort()willbecalledinitially andthenstage->inc()willbecalledforeachrecordprocessed.
@return DB_SUCCESS or error code */
dberr_t
row_merge_sort(
trx_t* trx, const row_merge_dup_t* dup,
merge_file_t* file,
row_merge_block_t* block,
pfs_os_file_t* tmpfd, constbool update_progress, /*!< in: update progress
status variable or not */ constdouble pct_progress, /*!< in: total progress percent
until now */ constdouble pct_cost, /*!< in: current progress percent */
row_merge_block_t* crypt_block, /*!< in: crypt buf or NULL */
ulint space, /*!< in: space id */
ut_stage_alter_t* stage)
{ const ulint half = file->offset / 2;
ulint num_runs;
ulint* run_offset;
dberr_t error = DB_SUCCESS;
ulint merge_count = 0;
ulint total_merge_sort_count; double curr_progress = 0;
DBUG_ENTER("row_merge_sort");
/* Record the number of merge runs we need to perform */
num_runs = file->offset;
if (stage != NULL) {
stage->begin_phase_sort(log2(double(num_runs)));
}
/* If num_runs are less than 1, nothing to merge */ if (num_runs <= 1) {
DBUG_RETURN(error);
}
for (ulint i = 0; i < dtuple_get_n_fields(tuple); i++) {
ulint len; constvoid* data;
dfield_t* field = dtuple_get_nth_field(tuple, i);
ulint field_len; const byte* field_data;
if (!dfield_is_ext(field)) { continue;
}
ut_ad(!dfield_is_null(field));
/* During the creation of a PRIMARY KEY, the table is X-locked,andweskipcopyingrecordsthathavebeen markedfordeletion.Therefore,externallystored columnscannotpossiblybefreedbetweenthetimethe BLOBpointersareread(row_merge_read_clustered_index())
and dereferenced (below). */ if (mrec == NULL) {
field_data
= static_cast<byte*>(dfield_get_data(field));
field_len = dfield_get_len(field);
data = btr_copy_externally_stored_field(
&len, field_data, zip_size, field_len, heap);
} else {
data = btr_rec_copy_externally_stored_field(
mrec, offsets, zip_size, i, &len, heap);
}
/* Because we have locked the table, any records writtenbyincompletetransactionsmusthavebeen rolledbackalready.Theremustnotbeanyincomplete
BLOB columns. */
ut_a(data);
dfield_set_data(field, data, len);
}
}
/** Convert a merge record to a typed data tuple. Note that externally storedfieldsarenotcopiedtoheap. @param[in,out]indexindexonthetable @param[in]mtuplemergerecord @param[in]heapmemoryheapfromwhichmemoryneededisallocated
@return index entry built. */ static void
row_merge_mtuple_to_dtuple(
dict_index_t* index,
dtuple_t* dtuple, const mtuple_t* mtuple)
{
memcpy(dtuple->fields, mtuple->fields,
dtuple->n_fields * sizeof *mtuple->fields);
}
if (row_buf != NULL) { if (n_rows >= row_buf->n_tuples) { break;
}
/* Convert merge tuple record from
row buffer to data tuple record */
row_merge_mtuple_to_dtuple(
index, dtuple, &row_buf->tuples[n_rows]);
n_rows++; /* BLOB pointers must be copied from dtuple */
mrec = NULL;
} else {
b = row_merge_read_rec(block, buf, b, index,
fd, &foffs, &mrec, offsets,
crypt_block,
space);
if (UNIV_UNLIKELY(!b)) { /* End of list, or I/O error */ if (mrec) {
error = DB_CORRUPTION;
} break;
}
Anymodificationsafterthe row_merge_read_clustered_index()scan
will go through row_log_table_apply(). */
row_merge_copy_blobs(
mrec, offsets,
old_table->space->zip_size(),
dtuple, tuple_heap);
}
/*********************************************************************//**
Drop an index that was created before an error occurred.
The data dictionary must have been locked exclusively by the caller,
because the transaction will not be committed. */ static void
row_merge_drop_index_dict( /*======================*/
trx_t* trx, /*!< in/out: dictionary transaction */
index_id_t index_id)/*!< in: index identifier */
{ staticconstchar sql[] = "PROCEDURE DROP_INDEX_PROC () IS\n" "BEGIN\n" "DELETE FROM SYS_FIELDS WHERE INDEX_ID=:indexid;\n" "DELETE FROM SYS_INDEXES WHERE ID=:indexid;\n" "END;\n";
dberr_t error;
pars_info_t* info;
info = pars_info_create();
pars_info_add_ull_literal(info, "indexid", index_id);
trx->op_info = "dropping index from dictionary";
error = que_eval_sql(info, sql, trx);
if (error != DB_SUCCESS) { /* Even though we ensure that DDL transactions are WAIT andDEADLOCKfree,wecouldencounterothererrorse.g.,
DB_TOO_MANY_CONCURRENT_TRXS. */
trx->error_state = DB_SUCCESS;
ib::error() << "row_merge_drop_index_dict failed with error "
<< error;
}
trx->op_info = "";
}
/*********************************************************************//**
Drop indexes that were created before an error occurred.
The data dictionary must have been locked exclusively by the caller,
because the transaction will not be committed. */ static void
row_merge_drop_indexes_dict( /*========================*/
trx_t* trx, /*!< in/out: dictionary transaction */
table_id_t table_id)/*!< in: table identifier */
{ staticconstchar sql[] = "PROCEDURE DROP_INDEXES_PROC () IS\n" "ixid CHAR;\n" "found INT;\n"
"DECLARE CURSOR index_cur IS\n" " SELECT ID FROM SYS_INDEXES\n" " WHERE TABLE_ID=:tableid AND\n" " SUBSTR(NAME,0,1)='" TEMP_INDEX_PREFIX_STR "'\n" "FOR UPDATE;\n"
"BEGIN\n" "found := 1;\n" "OPEN index_cur;\n" "WHILE found = 1 LOOP\n" " FETCH index_cur INTO ixid;\n" " IF (SQL % NOTFOUND) THEN\n" " found := 0;\n" " ELSE\n" " DELETE FROM SYS_FIELDS WHERE INDEX_ID=ixid;\n" " DELETE FROM SYS_INDEXES WHERE CURRENT OF index_cur;\n" " END IF;\n" "END LOOP;\n" "CLOSE index_cur;\n"
/* It is possible that table->n_ref_count > 1 when locked=TRUE.Inthiscase,allcodethatshouldhaveanopen handletothetablebewaitingforthenextstatementtoexecute, orwaitingforameta-datalock.
A concurrent purge will be prevented by dict_sys.latch. */
switch (error) { case DB_SUCCESS: break; default: /* Even though we ensure that DDL transactions are WAIT andDEADLOCKfree,wecouldencounterothererrorse.g.,
DB_TOO_MANY_CONCURRENT_TRXS. */
ib::error() << "row_merge_drop_indexes_dict failed with error "
<< error; /* fall through */ case DB_TOO_MANY_CONCURRENT_TRXS:
trx->error_state = DB_SUCCESS;
}
trx->op_info = "";
}
/** Drop common internal tables if all fulltext indexes are dropped @paramtrxtransaction
@param table user table */ staticvoid row_merge_drop_fulltext_indexes(trx_t *trx, dict_table_t *table)
{ if (DICT_TF2_FLAG_IS_SET(table, DICT_TF2_FTS_HAS_DOC_ID) ||
!table->fts ||
!ib_vector_is_empty(table->fts->indexes)) return;
for (const dict_index_t *index= dict_table_get_first_index(table);
index; index= dict_table_get_next_index(index)) if (index->type & DICT_FTS) return;
index = dict_table_get_first_index(table);
ut_ad(dict_index_is_clust(index));
ut_ad(dict_index_get_online_status(index) == ONLINE_INDEX_COMPLETE);
/* the caller should have an open handle to the table */
ut_ad(table->get_ref_count() >= 1);
/* It is possible that table->n_ref_count > 1 when locked=TRUE.Inthiscase,allcodethatshouldhaveanopen handletothetablebewaitingforthenextstatementtoexecute, orwaitingforameta-datalock.
A concurrent purge will be prevented by MDL. */
if (!locked && (table->get_ref_count() > 1
|| table->has_lock_other_than(alter_trx))) { while ((index = dict_table_get_next_index(index)) != NULL) {
ut_ad(!dict_index_is_clust(index));
switch (dict_index_get_online_status(index)) { case ONLINE_INDEX_ABORTED_DROPPED: continue; case ONLINE_INDEX_COMPLETE: if (index->is_committed()) { /* Do nothing to already
published indexes. */
} elseif (index->type & DICT_FTS) { /* Drop a completed FULLTEXT index,duetoatimeoutduring MDLupgradefor commit_inplace_alter_table(). Becauseonlyconcurrentreads areallowed(andtheyarenot seeingthisindexyet)we
are safe to drop the index. */
dict_index_t* prev = UT_LIST_GET_PREV(
indexes, index); /* At least there should be theclusteredindexbefore
this one. */
ut_ad(prev);
ut_a(table->fts);
fts_drop_index(table, index, trx);
row_merge_drop_index_dict(
trx, index->id); /* We can remove a DICT_FTS indexfromthecache,because wedonotallowADDFULLTEXTINDEX withLOCK=NONE.Ifweallowedthat, weshouldexcludeFTSentriesfrom prebuilt->ins_node->entry_list
in ins_node_create_entry_list(). */ #ifdef BTR_CUR_HASH_ADAPT
ut_ad(!index->search_info.ref_count); #endif/* BTR_CUR_HASH_ADAPT */
dict_index_remove_from_cache(
table, index);
index = prev;
} else {
index->lock.x_lock(SRW_LOCK_CALL);
dict_index_set_online_status(
index, ONLINE_INDEX_ABORTED);
index->type |= DICT_CORRUPT;
table->drop_aborted = TRUE; goto drop_aborted;
} continue; case ONLINE_INDEX_CREATION:
index->lock.x_lock(SRW_LOCK_CALL);
ut_ad(!index->is_committed());
row_log_abort_sec(index);
drop_aborted:
index->lock.x_unlock();
DEBUG_SYNC_C("merge_drop_index_after_abort"); /* covered by dict_sys.latch */
MONITOR_INC(MONITOR_BACKGROUND_DROP_INDEX); /* fall through */ case ONLINE_INDEX_ABORTED: /* Drop the index tree from the datadictionaryandfreeitfrom thetablespace,butkeeptheobject
in the data dictionary cache. */
row_merge_drop_index_dict(trx, index->id);
index->lock.x_lock(SRW_LOCK_CALL);
dict_index_set_online_status(
index, ONLINE_INDEX_ABORTED_DROPPED);
index->lock.x_unlock();
table->drop_aborted = TRUE; continue;
}
ut_error;
}
/* Invalidate all row_prebuilt_t::ins_graph that are referring tothistable.Thatis,forcerow_get_prebuilt_insert_row()to
rebuild prebuilt->ins_node->entry_list). */ if (table->def_trx_id < trx->id) {
table->def_trx_id = trx->id;
} else {
ut_ad(table->def_trx_id == trx->id || table->name.part());
}
next_index = dict_table_get_next_index(index);
while ((index = next_index) != NULL) { /* read the next pointer before freeing the index */
next_index = dict_table_get_next_index(index);
ut_ad(!dict_index_is_clust(index));
if (!index->is_committed()) { /* If it is FTS index, drop from table->fts
and also drop its auxiliary tables */ if (index->type & DICT_FTS) {
ut_a(table->fts);
fts_drop_index(table, index, trx);
}
switch (dict_index_get_online_status(index)) { case ONLINE_INDEX_CREATION: /* This state should only be possible whenprepare_inplace_alter_table()fails afterinvokingrow_merge_create_index(). Ininplace_alter_table(), row_merge_build_indexes() shouldneverleavetheindexinthisstate. Itwouldinvokerow_log_abort_sec()on
failure. */ case ONLINE_INDEX_COMPLETE: /* In these cases, we are able to drop theindexstraight.TheDROPINDEXwas
never deferred. */ break; case ONLINE_INDEX_ABORTED: case ONLINE_INDEX_ABORTED_DROPPED: /* covered by dict_sys.latch */
MONITOR_DEC(MONITOR_BACKGROUND_DROP_INDEX);
}
"DECLARE CURSOR cur_tab IS\n" "SELECT ID FROM SYS_TABLES\n" "WHERE INSTR(NAME,:name)+45=LENGTH(NAME)" " AND INSTR('123456',SUBSTR(NAME,LENGTH(NAME)-1,1))>0" " FOR UPDATE;\n"
"DECLARE CURSOR cur_idx IS\n" "SELECT ID FROM SYS_INDEXES\n" "WHERE TABLE_ID = tid FOR UPDATE;\n"
"BEGIN\n" "OPEN cur_tab;\n" "WHILE 1 = 1 LOOP\n" " FETCH cur_tab INTO tid;\n" " IF (SQL % NOTFOUND) THEN EXIT; END IF;\n" " OPEN cur_idx;\n" " WHILE 1 = 1 LOOP\n" " FETCH cur_idx INTO iid;\n" " IF (SQL % NOTFOUND) THEN EXIT; END IF;\n" " DELETE FROM SYS_FIELDS WHERE INDEX_ID=iid;\n" " DELETE FROM SYS_INDEXES WHERE CURRENT OF cur_idx;\n" " END LOOP;\n" " CLOSE cur_idx;\n" " DELETE FROM SYS_COLUMNS WHERE TABLE_ID=tid;\n" " DELETE FROM SYS_TABLES WHERE CURRENT OF cur_tab;\n" "END LOOP;\n" "CLOSE cur_tab;\n" "END;\n";
/** During recovery, drop recovered index stubs that were created in
prepare_inplace_alter_table_dict(). */ void row_merge_drop_temp_indexes()
{
static_assert(DICT_FTS == 32, "compatibility");
"DECLARE CURSOR fts_cur IS\n" " SELECT TABLE_ID,ID FROM SYS_INDEXES\n" " WHERE TYPE=32" " AND SUBSTR(NAME,0,1)='" TEMP_INDEX_PREFIX_STR "'\n" " FOR UPDATE;\n"
"DECLARE CURSOR index_cur IS\n" " SELECT ID FROM SYS_INDEXES\n" " WHERE SUBSTR(NAME,0,1)='" TEMP_INDEX_PREFIX_STR "'\n" "FOR UPDATE;\n"
"BEGIN\n" "found := 1;\n" "OPEN fts_cur;\n" "WHILE found = 1 LOOP\n" " FETCH fts_cur INTO drop_fts();\n" " IF (SQL % NOTFOUND) THEN\n" " found := 0;\n" " END IF;\n" "END LOOP;\n" "CLOSE fts_cur;\n"
"OPEN index_cur;\n" "WHILE found = 1 LOOP\n" " FETCH index_cur INTO ixid;\n" " IF (SQL % NOTFOUND) THEN\n" " found := 0;\n" " ELSE\n" " DELETE FROM SYS_FIELDS WHERE INDEX_ID=ixid;\n" " DELETE FROM SYS_INDEXES WHERE CURRENT OF index_cur;\n" " END IF;\n" "END LOOP;\n" "CLOSE index_cur;\n" "END;\n";
/* Load the table definitions that contain partially defined indexes,sothatthedatadictionaryinformationcanbechecked
when accessing the tablename.ibd files. */
trx_t* trx = trx_create();
trx_start_for_ddl(trx);
trx->op_info = "dropping partially created indexes";
dberr_t error = lock_sys_tables(trx);
row_mysql_lock_data_dictionary(trx); /* Ensure that this transaction will be rolled back and locks willbereleased,iftheservergetskilledbeforethecommit
gets written to the redo log. */
trx->dict_operation = true;
if (error) { /* Even though we ensure that DDL transactions are WAIT andDEADLOCKfree,wecouldencounterothererrorse.g.,
DB_TOO_MANY_CONCURRENT_TRXS. */
trx->error_state = DB_SUCCESS;
/** Create temporary merge files in the given parameter path, and if UNIV_PFS_IOisdefined,registerthefiledescriptorwithPerformanceSchema. @param[in]pathlocationforcreatingtemporarymergefiles,orNULL
@return File descriptor */ static pfs_os_file_t row_merge_file_create_mode(constchar *path, int mode)
{ if (!path) {
path = mysql_tmpdir;
} #ifdef UNIV_PFS_IO /* This temp file open does not go through normal fileAPIs,addinstrumentationtoregisterwith
performance schema */ struct PSI_file_locker* locker;
PSI_file_locker_state state; staticconstchar label[] = "/Innodb Merge Temp File"; char* name = static_cast<char*>(
ut_malloc_nokey(strlen(path) + sizeof label));
strcpy(name, path);
strcat(name, label);
/*********************************************************************//**
Rename an index in the dictionary that was created. The data
dictionary must have been locked exclusively by the caller, because
the transaction will not be committed.
@return DB_SUCCESS if all OK */
dberr_t
row_merge_rename_index_to_add( /*==========================*/
trx_t* trx, /*!< in/out: transaction */
table_id_t table_id, /*!< in: table identifier */
index_id_t index_id) /*!< in: index identifier */
{
dberr_t err = DB_SUCCESS;
pars_info_t* info = pars_info_create();
/* We use the private SQL parser of Innobase to generate the
query graphs needed in renaming indexes. */
staticconstchar rename_index[] = "PROCEDURE RENAME_INDEX_PROC () IS\n" "BEGIN\n" "UPDATE SYS_INDEXES SET NAME=SUBSTR(NAME,1,LENGTH(NAME)-1)\n" "WHERE TABLE_ID = :tableid AND ID = :indexid;\n" "END;\n";
if (err != DB_SUCCESS) { /* Even though we ensure that DDL transactions are WAIT andDEADLOCKfree,wecouldencounterothererrorse.g.,
DB_TOO_MANY_CONCURRENT_TRXS. */
trx->error_state = DB_SUCCESS;
/** Create the index and load in to the dictionary. @param[in,out]tabletheindexisonthistable @param[in]index_deftheindexdefinition @param[in]add_vnewvirtualcolumnsaddedalongwithadd indexcall
@return index, or NULL on error */
dict_index_t*
row_merge_create_index(
dict_table_t* table, const index_def_t* index_def, const dict_add_v_col_t* add_v)
{
dict_index_t* index;
ulint n_fields = index_def->n_fields;
ulint i;
ulint n_add_vcol = 0;
DBUG_ENTER("row_merge_create_index");
ut_ad(!srv_read_only_mode);
ut_ad(!recv_sys.rpo);
/* Create the index prototype, using the passed in def, this is not apersistentoperation.Wepass0asthespaceid,anddetermineat
a lower level the space id where to store the table. */
index = dict_mem_index_create(table, index_def->name,
index_def->ind_type, n_fields);
index->set_committed(index_def->rebuild);
for (i = 0; i < n_fields; i++) { constchar* name;
index_field_t* ifield = &index_def->fields[i];
if (ifield->is_v_col) { if (ifield->col_no >= table->n_v_def) {
ut_ad(ifield->col_no < table->n_v_def
+ add_v->n_v_col);
ut_ad(ifield->col_no >= table->n_v_def);
name = add_v->v_col_name[
ifield->col_no - table->n_v_def];
n_add_vcol++;
} else {
name = dict_table_get_v_col_name(
table, ifield->col_no).str;
}
} else {
name = dict_table_get_col_name(table, ifield->col_no).str;
}
if (n_add_vcol) {
index->assign_new_v_col(n_add_vcol);
}
DBUG_RETURN(index);
}
/*********************************************************************//**
Check if a transaction can use an index. */ bool
row_merge_is_index_usable( /*======================*/ const trx_t* trx, /*!< in: transaction */ const dict_index_t* index) /*!< in: index to check */
{ if (!index->is_primary()
&& dict_index_is_online_ddl(index)) { /* Indexes that are being created are not useable. */ return(false);
}
/* This will allocate "3 * srv_sort_buf_size" elements of type
row_merge_block_t. The latter is defined as byte. */
block_size = 3 * srv_sort_buf_size;
block = alloc.allocate_large(block_size, &block_pfx);
if (block == NULL) {
DBUG_RETURN(DB_OUT_OF_MEMORY);
}
/* Initialize all the merge file descriptors, so that we don'tcallrow_merge_file_destroy()onuninitialized
merge file descriptor */
for (i = 0; i < n_merge_files; i++) {
merge_files[i].fd = OS_FILE_CLOSED;
merge_files[i].offset = 0;
merge_files[i].n_rec = 0;
}
total_static_cost = COST_BUILD_INDEX_STATIC
* static_cast<double>(n_indexes) + COST_READ_CLUSTERED_INDEX;
total_dynamic_cost = COST_BUILD_INDEX_DYNAMIC
* static_cast<double>(n_indexes); for (i = 0; i < n_indexes; i++) { if (indexes[i]->type & DICT_FTS) { bool opt_doc_id_size = false;
/* To build FTS index, we would need to extract doc'sword,DocID,andword'sposition,so weneedtobuilda"ftssortindex"indexing
on above three 'fields' */
/* Check whether we can use 4 bytes instead of 8 bytes integerfieldtoholdtheDocID,thusreduce theoverallsortsize.Ifthefulltextindexisbeing addedforthefirsttimethenweshoulduse8bytesDocID sizebecausetable->stat_n_rowsisanestimationand
not reliable to determine the Doc ID size. */ if (old_table->fts || !DICT_TF2_FLAG_IS_SET(
new_table, DICT_TF2_FTS_ADD_DOC_ID)) { /* If the Doc ID column is supplied by user orrebuildingtheexistingFTStable,then
check the maximum Doc ID in the old table */
doc_id_t max_doc_id =
fts_get_max_doc_id((dict_table_t*) old_table);
opt_doc_id_size =
(max_doc_id < MAX_DOC_ID_OPT_VAL);
}
/* This can fail e.g. if temporal files can't be
created */ if (!row_fts_psort_info_init(
trx, dup, new_table, opt_doc_id_size,
old_table->space->zip_size(),
&psort_info, &merge_info)) {
error = DB_CORRUPTION; goto func_exit;
}
/* We need to ensure that we free the resources
allocated */
fts_psort_initiated = true;
}
}
if (global_system_variables.log_warnings > 2) {
sql_print_information("InnoDB: Online DDL : Start reading" " clustered index of the table" " and create temporary files");
}
/* Do not continue if we can't encrypt table pages */ if (!old_table->is_readable() ||
!new_table->is_readable()) {
error = innodb_decryption_failed(trx->mysql_thd,
!old_table->is_readable()
? old_table : new_table); goto func_exit;
}
/* Read clustered index of the table and create files for
secondary index entries for merge sort */
error = row_merge_read_clustered_index(
trx, table, old_table, new_table, online, indexes,
fts_sort_idx, psort_info, merge_files, key_numbers,
n_indexes, defaults, add_v, col_map, add_autoinc,
sequence, block, skip_pk_sort, &tmpfd, stage,
pct_cost, crypt_block, eval_table, allow_not_null,
col_collate);
stage->end_phase_read_pk();
pct_progress += pct_cost;
if (global_system_variables.log_warnings > 2) {
sql_print_information("InnoDB: Online DDL : End of reading " "clustered index of the table" " and create temporary files");
}
for (i = 0; i < n_merge_files; i++) {
total_index_blocks += merge_files[i].offset;
}
if (error != DB_SUCCESS) { goto func_exit;
}
DEBUG_SYNC_C("row_merge_after_scan");
/* Now we have files containing index entries ready for
sorting and inserting. */
for (ulint k = 0, i = 0; i < n_indexes; i++) {
dict_index_t* sort_idx = indexes[i];
if (dict_index_is_spatial(sort_idx)) { continue;
}
if (indexes[i]->type & DICT_FTS) {
sort_idx = fts_sort_idx;
if (FTS_PLL_MERGE) {
row_fts_start_parallel_merge(merge_info); for (j = 0; j < FTS_NUM_AUX_INDEX; j++) {
merge_info[j].task->wait(); delete merge_info[j].task;
}
} else { /* This cannot report duplicates; an
assertion would fail in that case. */
error = row_fts_merge_insert(
sort_idx, new_table,
psort_info, 0);
}
if (online && old_table == new_table && error != DB_SUCCESS) { /* On error, flag all online secondary index creation
as aborted. */ for (i = 0; i < n_indexes; i++) {
ut_ad(!(indexes[i]->type & DICT_FTS));
ut_ad(!indexes[i]->is_committed());
ut_ad(!dict_index_is_clust(indexes[i]));
/* Completed indexes should be dropped as well,andindexeswhosecreationwasaborted shouldbedroppedfromthepersistent storage.However,atthispointwecanonly setsomeflagsinthenot-yet-published indexes.Theseindexeswillbedroppedlater inrow_merge_drop_indexes(),calledby
rollback_inplace_alter_table(). */
switch (dict_index_get_online_status(indexes[i])) { case ONLINE_INDEX_COMPLETE: break; case ONLINE_INDEX_CREATION:
indexes[i]->lock.x_lock(SRW_LOCK_CALL);
row_log_abort_sec(indexes[i]);
indexes[i]->type |= DICT_CORRUPT;
indexes[i]->lock.x_unlock();
new_table->drop_aborted = TRUE; /* fall through */ case ONLINE_INDEX_ABORTED_DROPPED: case ONLINE_INDEX_ABORTED:
MONITOR_ATOMIC_INC(
MONITOR_BACKGROUND_DROP_INDEX);
}
}
if (buf->n_tuples == 0)
{ /* Tuple data size is greater than srv_sort_buf_size */
ut_ad(i == 0); if (!large_tuple_heap)
large_tuple_heap= mem_heap_create(DTUPLE_EST_ALLOC(row.n_fields));
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.