/**************************************************//**
@file btr/btr0cur.cc
The index tree cursor
All changes that row operations make to a B-tree or the records
there must go through this module! Undo log records are written here
of every modify or insert of a clustered index record.
NOTE!!!
To make sure we donot run out of disk space during a pessimistic
insert or update, we have to reserve 2 x the height of the index tree
many pages in the tablespace before we start the operation, because if leaf splitting has been started, it is difficult to undo, except
by crashing the database and doing a roll-forward.
Created 10/16/1994 Heikki Tuuri
*******************************************************/
/** Modification types for the B-tree operation. NotethattheordermustbeDELETE,BOTH,INSERT!!
*/ enum btr_intention_t {
BTR_INTENTION_DELETE,
BTR_INTENTION_BOTH,
BTR_INTENTION_INSERT
};
/** For the index->lock scalability improvement, only possibility of clear performanceregressionobservedwascausedbygrownhugehistorylistlength. Thatisbecausetheexclusiveuseofindex->lockalsoworkedasreserving freeblocksandreadIObandwidthwithpriority.Toavoidhugeglowinghistory listassamelevelwithpreviousimplementation,prioritizespessimistictree operationsbypurgeastheprevious,whenitseemstobegrowinghuge.
Experimentally,thehistorylistlengthstartstoaffecttoperformance
throughput clearly from about 100000. */ #define BTR_CUR_FINE_HISTORY_LENGTH 100000
#ifdef UNIV_DEBUG /* Flag to limit optimistic insert records */
uint btr_cur_limit_optimistic_insert_debug; /** Number of times index lock was upgraded from SX to X */
Atomic_counter<uint64_t> btr_cur_n_index_lock_upgrades{0}; /** Number of times btr_cur_pessimistic_insert() was called */
Atomic_counter<uint64_t> btr_cur_pessimistic_insert_calls{0}; /** Number of times btr_cur_pessimistic_update() was called */
Atomic_counter<uint64_t> btr_cur_pessimistic_update_calls{0}; /** Number of times btr_cur_pessimistic_delete() was called */
Atomic_counter<uint64_t> btr_cur_pessimistic_delete_calls{0}; /** Number of times DB_UNDERFLOW was returned as optimistic update error in btr_cur_pessimistic_update() */
Atomic_counter<uint64_t> btr_cur_pessimistic_update_optim_err_underflows{0}; /** Number of times DB_OVERFLOW was returned as optimistic update error in btr_cur_pessimistic_update() */
Atomic_counter<uint64_t> btr_cur_pessimistic_update_optim_err_overflows{0}; #endif/* UNIV_DEBUG */
/** innodb_index_shrink: whether InnoDB may shrink a B-tree by merging or reorganizingpagesonarecord-growingUPDATE.ON(thedefault)keepsthe historicalbehavior.WhenOFF,btr_cur_optimistic_update()gatesits DB_UNDERFLOWreturnbehindactualrecordshrinkage(soafreshlysplitpageis notre-mergedjustbecauseanupdategrewarecord),and btr_cur_pessimistic_update()skipsbtr_cur_insert_if_possible()onthe DB_OVERFLOWfallbackforagrowingrecordwhenanuncompressedpagecannot satisfyBTR_CUR_PAGE_REORGANIZE_LIMITafterareorganize,fallingthroughtoa pagesplitinstead(onlysize-growingUPDATEspaythecostofanearlypage
split). */
my_bool btr_cur_index_shrink;
/** In the optimistic insert, if the insert does not fit, but this much space
can be released by page reorganize, then it is reorganized */ #define BTR_CUR_PAGE_REORGANIZE_LIMIT (srv_page_size / 32)
/** The structure of a BLOB part header */ /* @{ */ /*--------------------------------------*/ #define BTR_BLOB_HDR_PART_LEN 0/*!< BLOB part len on this
page */ #define BTR_BLOB_HDR_NEXT_PAGE_NO 4/*!< next BLOB part page no,
FIL_NULL if none */ /*--------------------------------------*/ #define BTR_BLOB_HDR_SIZE 8/*!< Size of a BLOB
part header, in bytes */
/* @} */
/*******************************************************************//**
Marks all extern fields in a record as owned by the record. This function
should be called if the delete mark of a record is removed: a notdelete
marked record always owns all its extern fields. */ static void
btr_cur_unmark_extern_fields( /*=========================*/
buf_block_t* block, /*!< in/out: index page */
rec_t* rec, /*!< in/out: record in a clustered index */
dict_index_t* index, /*!< in: index of the page */ const rec_offs* offsets,/*!< in: array returned by rec_get_offsets() */
mtr_t* mtr); /*!< in: mtr, or NULL if not logged */ /***********************************************************//**
Frees the externally stored fields for a record, if the field is mentioned
in the update vector. */ static void
btr_rec_free_updated_extern_fields( /*===============================*/
dict_index_t* index, /*!< in: index of rec; the index tree MUST be
X-latched */
rec_t* rec, /*!< in: record */
buf_block_t* block, /*!< in: index page of rec */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec, index) */ const upd_t* update, /*!< in: update vector */ bool rollback,/*!< in: performing rollback? */
mtr_t* mtr); /*!< in: mini-transaction handle which contains
an X-latch to record page and to the tree */ /***********************************************************//**
Frees the externally stored fields for a record. */ static void
btr_rec_free_externally_stored_fields( /*==================================*/
dict_index_t* index, /*!< in: index of the data, the index
tree MUST be X-latched */
rec_t* rec, /*!< in: record */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec, index) */
buf_block_t* block, /*!< in: index page of rec */ bool rollback,/*!< in: performing rollback? */
mtr_t* mtr); /*!< in: mini-transaction handle which contains anX-latchtorecordpageandtotheindex
tree */
/** Load the instant ALTER TABLE metadata from the clustered index whenloadingatabledefinition. @param[in,out]indexclusteredindexdefinition @param[in,out]mtrmini-transaction @returnerrorcode @retvalDB_SUCCESSifnoerroroccurred
@retval DB_CORRUPTION if any corruption was noticed */ static dberr_t btr_cur_instant_init_low(dict_index_t* index, mtr_t* mtr)
{
ut_ad(index->is_primary());
ut_ad(index->table->is_readable());
if (!index->table->supports_instant()) { return DB_SUCCESS;
}
if (page_rec_is_supremum(rec)
|| !(info_bits & REC_INFO_MIN_REC_FLAG)) { if (rec && !index->is_instant()) { /* The FIL_PAGE_TYPE_INSTANT and PAGE_INSTANT may be assignedevenifinstantADDCOLUMNwasnot committed.Changestothesepageheaderfieldsarenot undo-logged,butchangestothehiddenmetadatarecord are.Iftheserveriskilledandrestarted,thepage headerfieldscouldremainseteventhoughnometadata
record is present. */ return DB_SUCCESS;
}
ib::error() << "Table " << index->table->name
<< " is missing instant ALTER metadata";
index->table->corrupted = true; return DB_CORRUPTION;
}
/* Read the metadata. We can get here on server restart orwhenthetablewasevictedfromthedatadictionarycache andisnowbeingaccessedagain.
Here,READCOMMITTEDandREPEATABLEREADshouldbeequivalent. CommittingtheADDCOLUMNoperationwouldacquire MDL_EXCLUSIVEandLOCK_X|LOCK_TABLE,whichwouldpreventany concurrentoperationsonthetable,includingtableeviction
from the cache. */
if (info_bits & REC_INFO_DELETED_FLAG) { /* This metadata record includes a BLOB that identifies
any dropped or reordered columns. */
ulint trx_id_offset = index->trx_id_offset; /* If !index->trx_id_offset, the PRIMARY KEY contains variable-lengthcolumns.Forthemetadatarecord, variable-lengthcolumnsshouldbewrittenwithzero length.However,beforeMDEV-21088wasfixed,for variable-lengthencodedPRIMARYKEYcolumnoftype CHAR,wewrotemorethanzerobytes.Thatiswhywe mustdeterminetheactuallengthofeachPRIMARYKEY column.TheDB_TRX_IDwillstartrightafterany
PRIMARY KEY columns. */
ut_ad(index->n_uniq);
/* We cannot invoke rec_get_offsets() before index->table->deserialise_columns().Therefore,
we must duplicate some logic here. */ if (trx_id_offset) {
} elseif (index->table->not_redundant()) { /* The PRIMARY KEY contains variable-length columns. Forthemetadatarecord,variable-lengthcolumnsare alwayswrittenwithzerolength.TheDB_TRX_IDwill
start right after any fixed-length columns. */
/* OK, before MDEV-21088 was fixed, for variable-lengthencodedPRIMARYKEYcolumnof typeCHAR,wewrotemorethanzerobytes.In ordertoallowaffectedtablestobeaccessed, itwouldbenicetodeterminetheactual lengthofeachPRIMARYKEYcolumn.However,to beabletodothat,weshoulddeterminethe sizeofthenull-bitbitmapinthemetadata record.Andwecannotknowthatbeforereading themetadataBLOB,whosestartingpointweare tryingtofindhere.(AlthoughthePRIMARYKEY columnscannotbeNULL,wewouldhavetoknow wherethelengthsofvariable-lengthPRIMARYKEY columnsstart.)
So,unfortunatelywecannothelpuserswho wereaffectedbyMDEV-21088onaROW_FORMAT=COMPACT
or ROW_FORMAT=DYNAMIC table. */
/* The unused part of the BLOB page should be zero-filled. */ for (const byte* b = block->page.frame
+ (FIL_PAGE_DATA + BTR_BLOB_HDR_SIZE) + len,
* const end = block->page.frame + srv_page_size
- BTR_EXTERN_LEN;
b < end; ) { if (*b++) { goto incompatible;
}
}
/* In fact, because we only ever append fields to the metadata record,itisalsoOKtoperformREADUNCOMMITTEDand thenignoreanyextrafields,providedthat
trx_sys.is_registered(DB_TRX_ID). */ if (rec_offs_n_fields(offsets)
> ulint(index->n_fields) + !!index->table->instant
&& !trx_sys.is_registered(current_trx(),
row_get_rec_trx_id(rec, index,
offsets))) { goto inconsistent;
}
for (unsigned i = index->n_core_fields; i < index->n_fields; i++) {
dict_col_t* col = index->fields[i].col; constunsigned o = i + !!index->table->instant;
ulint len; const byte* data = rec_get_nth_field(rec, offsets, o, &len);
ut_ad(!col->is_added());
ut_ad(!col->def_val.data);
col->def_val.len = len; switch (len) { case UNIV_SQL_NULL: continue; case0:
col->def_val.data = field_ref_zero; continue;
}
ut_ad(len != UNIV_SQL_DEFAULT); if (!rec_offs_nth_extern(offsets, o)) {
col->def_val.data = mem_heap_dup(
index->table->heap, data, len);
} elseif (len < BTR_EXTERN_FIELD_REF_SIZE
|| !memcmp(data + len - BTR_EXTERN_FIELD_REF_SIZE,
field_ref_zero,
BTR_EXTERN_FIELD_REF_SIZE)) {
col->def_val.len = UNIV_SQL_DEFAULT; goto inconsistent;
} else {
col->def_val.data = btr_copy_externally_stored_field(
&col->def_val.len, data,
cur.page_cur.block->zip_size(),
len, index->table->heap);
}
}
mem_heap_free(heap); return DB_SUCCESS;
}
/** Load the instant ALTER TABLE metadata from the clustered index whenloadingatabledefinition. @param[in,out]mtrmini-transaction @param[in,out]tabletabledefinitionfromthedatadictionary @returnerrorcode
@retval DB_SUCCESS if no error occurred */
dberr_t btr_cur_instant_init(mtr_t *mtr, dict_table_t *table)
{
dict_index_t *index= dict_table_get_first_index(table);
mtr->start();
dberr_t err= index ? btr_cur_instant_init_low(index, mtr) : DB_CORRUPTION;
mtr->commit(); if (err == DB_SUCCESS && index->is_gen_clust())
{
btr_cur_t cur;
mtr->start();
err= cur.open_leaf(false, index, BTR_SEARCH_LEAF, mtr); if (err != DB_SUCCESS); elseif (const rec_t *rec= page_rec_get_prev(btr_cur_get_rec(&cur))) if (page_rec_is_user_rec(rec))
table->row_id= mach_read_from_6(rec);
mtr->commit();
} return err;
}
/** Initialize the n_core_null_bytes on first access to a clustered indexrootpage. @param[in]indexclusteredindexthatisonitsfirstaccess @param[in]pageclusteredindexrootpage
@return whether the page is corrupted */ bool btr_cur_instant_root_init(dict_index_t* index, const page_t* page)
{
ut_ad(!index->is_dummy);
ut_ad(index->is_primary());
ut_ad(!index->is_instant());
ut_ad(index->table->supports_instant());
if (page_has_siblings(page)) { returntrue;
}
/* This is normally executed as part of btr_cur_instant_init() whendict_load_table_one()isloadingatabledefinition. Otherthreadsshouldnotaccessormodifythen_core_null_bytes, n_core_fieldsbeforedict_load_table_one()returns.
ThiscanalsobeexecutedduringIMPORTTABLESPACE,wherethe
table definition is exclusively locked. */
switch (fil_page_get_type(page)) { default: returntrue; case FIL_PAGE_INDEX: /* The field PAGE_INSTANT is guaranteed 0 on clustered indexrootpagesofROW_FORMAT=COMPACTor
ROW_FORMAT=DYNAMIC when instant ADD COLUMN is not used. */ if (page_is_comp(page) && page_get_instant(page)) { returntrue;
}
index->n_core_null_bytes = static_cast<uint8_t>(
UT_BITS_IN_BYTES(unsigned(index->n_nullable))); returnfalse; case FIL_PAGE_TYPE_INSTANT: break;
}
const uint16_t n = page_get_instant(page);
if (n < index->n_uniq + DATA_ROLL_PTR) { /* The PRIMARY KEY (or hidden DB_ROW_ID) and DB_TRX_ID,DB_ROLL_PTRcolumnsmustalwaysbepresent
as 'core' fields. */ returntrue;
}
if (n > REC_MAX_N_FIELDS) { returntrue;
}
index->n_core_fields = n & dict_index_t::MAX_N_FIELDS;
if (!memcmp(infimum, "infimum", 8)
&& !memcmp(supremum, "supremum", 8)) { if (n > index->n_fields) { /* All fields, including those for instantly addedcolumns,mustbepresentinthe
data dictionary. */ returntrue;
}
if (memcmp(infimum, field_ref_zero, 8)
|| memcmp(supremum, field_ref_zero, 7)) { /* The infimum and supremum records must either contain theoriginalstrings,ortheymustbefilledwithzero
bytes, except for the bytes that we have repurposed. */ returntrue;
}
/** Getsintentioninbtr_intention_tfromlatch_mode,andclearestheintention atthelatch_mode. @paramlatch_modein/out:pointertolatch_mode
@return intention for latching tree */ static
btr_intention_t btr_cur_get_and_clear_intention(btr_latch_mode *latch_mode)
{
btr_intention_t intention;
switch (*latch_mode & (BTR_LATCH_FOR_INSERT | BTR_LATCH_FOR_DELETE)) { case BTR_LATCH_FOR_INSERT:
intention = BTR_INTENTION_INSERT; break; case BTR_LATCH_FOR_DELETE:
intention = BTR_INTENTION_DELETE; break; default: /* both or unknown */
intention = BTR_INTENTION_BOTH;
}
*latch_mode = btr_latch_mode(
*latch_mode & ~(BTR_LATCH_FOR_INSERT | BTR_LATCH_FOR_DELETE));
return(intention);
}
/** @return whether the distance between two records is at most the
specified value */ template<bool comp> staticbool
page_rec_distance_is_at_most(const page_t *page, const rec_t *left, const rec_t *right, ulint val)
noexcept
{ do
{ if (left == right) returntrue;
left= page_rec_next_get<comp>(page, left);
} while (left && val--); returnfalse;
}
/** Detects whether the modifying record might need a modifying tree structure. @param[in]indexindex @param[in]pagepage @param[in]lock_intentionlockintentionforthetreeoperation @param[in]recrecord(currentnode_ptr) @param[in]rec_sizesizeoftherecordormaxsizeofnode_ptr @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]mtrmtr
@return true if tree modification is needed */ static bool
btr_cur_will_modify_tree(
dict_index_t* index, const page_t* page,
btr_intention_t lock_intention, const rec_t* rec,
ulint rec_size,
ulint zip_size,
mtr_t* mtr)
{
ut_ad(!page_is_leaf(page));
ut_ad(mtr->memo_contains_flagged(&index->lock, MTR_MEMO_X_LOCK
| MTR_MEMO_SX_LOCK));
/* Pessimistic delete of the first record causes delete & insert ofnode_ptratupperlevel.Andasubsequentpageshrinkis possible.Itcausesdeleteofnode_ptrattheupperlevel. Soweshouldpayattentionalsoto2ndrecordnotonly firstrecordandlastrecord.Becauseifthe"delete&insert"are doneforthedifferentpage,the2ndrecordbecome firstrecordandfollowingcompressmightdeletetherecordandcauses
the uppper level node_ptr modification. */
if (lock_intention == BTR_INTENTION_BOTH) {
ulint level = btr_page_get_level(page);
/* This value is the worst expectation for the node_ptr recordstobedeletedfromthispage.Itisusedto expectwhetherthecursorpositioncanbetheleft_most
record in this page or not. */
ulint max_nodes_deleted = 0;
/* By modifying tree operations from the under of this level,logically(2^(level-1))opportunitiesto
deleting records in maximum even unreally rare case. */ if (level > 7) { /* TODO: adjust this practical limit. */
max_nodes_deleted = 64;
} elseif (level > 0) {
max_nodes_deleted = (ulint)1 << (level - 1);
} /* check delete will cause. (BTR_INTENTION_BOTH
or BTR_INTENTION_DELETE) */ if (n_recs <= max_nodes_deleted * 2) { /* The cursor record can be the left most record
in this page. */ returntrue;
}
/* Delete at leftmost record in a page causes delete &insertatitsparentpage.Afterthat,thedelete mightcausebtr_compress()anddeleterecordatits
parent page. Thus we should consider max deletes. */
margin *= max_nodes_deleted;
}
/* Safe because we already have SX latch of the index tree */ if (page_get_data_size(page)
< margin + BTR_CUR_PAGE_COMPRESS_LIMIT(index)) { return(true);
}
}
if (lock_intention >= BTR_INTENTION_BOTH) { /* check insert will cause. BTR_INTENTION_BOTH
or BTR_INTENTION_INSERT*/
/* Once we invoke the btr_cur_limit_optimistic_insert_debug, weshouldcheckithereinadvance,sincethemaxallowable
records in a page is limited. */
LIMIT_OPTIMISTIC_INSERT_DEBUG(n_recs, returntrue);
/* needs 2 records' space for the case the single split and insertcannotfit. page_get_max_insert_size_after_reorganize()includesspace
for page directory already */
ulint max_size
= page_get_max_insert_size_after_reorganize(page, 2);
/* TODO: optimize this condition for ROW_FORMAT=COMPRESSED. Thisisbasedontheworstcase,andwecouldinvoke
page_zip_available() on the block->page.zip. */ /* needs 2 records' space also for worst compress rate. */ if (zip_size
&& page_zip_empty_size(index->n_fields, zip_size)
<= rec_size * 2 + page_get_data_size(page)
+ page_dir_calc_reserved_space(n_recs + 2)) { return(true);
}
}
return(false);
}
/** Detects whether the modifying record might need a opposite modification totheintention. @parambpagebufferpoolpage @paramis_clustwhetherthisisaclusteredindex @paramlock_intentionlockintentionforthetreeoperation @paramnode_ptr_max_sizethemaximumsizeofanodepointer @paramcompress_limitBTR_CUR_PAGE_COMPRESS_LIMIT(index) @paramrecrecord(currentnode_ptr)
@return true if tree modification is needed */ staticbool btr_cur_need_opposite_intention(const buf_page_t &bpage, bool is_clust,
btr_intention_t lock_intention,
ulint node_ptr_max_size,
ulint compress_limit, const rec_t *rec)
{
ut_ad(bpage.frame == page_align(rec)); if (UNIV_LIKELY_NULL(bpage.zip.data) &&
!page_zip_available(&bpage.zip, is_clust, node_ptr_max_size, 1)) returntrue; const page_t *const page= bpage.frame; if (lock_intention != BTR_INTENTION_INSERT)
{ /* We compensate also for btr_cur_compress_recommendation() */ if (!page_has_siblings(page) ||
page_rec_is_first(rec, page) || page_rec_is_last(rec, page) ||
page_get_data_size(page) < node_ptr_max_size + compress_limit) returntrue; if (lock_intention == BTR_INTENTION_DELETE) returnfalse;
} elseif (page_has_next(page) && page_rec_is_last(rec, page)) returntrue;
LIMIT_OPTIMISTIC_INSERT_DEBUG(page_get_n_recs(page), returntrue); const ulint max_size= page_get_max_insert_size_after_reorganize(page, 2); return max_size < BTR_CUR_PAGE_REORGANIZE_LIMIT + node_ptr_max_size ||
max_size < node_ptr_max_size * 2;
}
/** @param[in]indexb-tree
@return maximum size of a node pointer record in bytes */ static ulint btr_node_ptr_max_size(const dict_index_t* index)
{ /* Each record has page_no, length of page_no and header. */
ulint comp = dict_table_is_comp(index->table);
ulint rec_max_size = comp
? REC_NODE_PTR_SIZE + 1 + REC_N_NEW_EXTRA_BYTES
+ UT_BITS_IN_BYTES(index->n_nullable)
: REC_NODE_PTR_SIZE + 2 + REC_N_OLD_EXTRA_BYTES
+ 2 * index->n_fields;
/* Compute the maximum possible record size. */ for (ulint i = 0; i < dict_index_get_n_unique_in_tree(index); i++) { const dict_field_t* field
= dict_index_get_nth_field(index, i); const dict_col_t* col
= dict_field_get_col(field);
ulint field_max_size;
ulint field_ext_max_size;
/* Determine the maximum length of the index field. */
field_max_size = dict_col_get_fixed_size(col, comp); if (field_max_size && field->fixed_len) { /* dict_index_add_col() should guarantee this */
ut_ad(!field->prefix_len
|| field->fixed_len == field->prefix_len); /* Fixed lengths are not encoded
in ROW_FORMAT=COMPACT. */
rec_max_size += field_max_size; continue;
}
field_max_size = dict_col_get_max_size(col); if (UNIV_UNLIKELY(!field_max_size)) { switch (col->mtype) { case DATA_VARCHAR: if (!comp
&& (!strcmp(index->table->name.m_name, "SYS_FOREIGN")
|| !strcmp(index->table->name.m_name, "SYS_FOREIGN_COLS"))) { break;
} /* fall through */ case DATA_FIXBINARY: case DATA_BINARY: case DATA_VARMYSQL: case DATA_CHAR: case DATA_MYSQL: /* BINARY(0), VARBINARY(0), CHAR(0)andVARCHAR(0)arepossible datatypedefinitionsinMariaDB. TheInnoDBinternalSQLparsermaps CHARtoDATA_VARCHAR,soDATA_CHAR(or DATA_MYSQL)isonlycomingfromthe
MariaDB SQL layer. */ if (comp) { /* Add a length byte, because fixed-lengthemptyfieldare encodedasvariable-length. ForROW_FORMAT=REDUNDANT, thesebyteswereaddedto
rec_max_size before this loop. */
rec_max_size++;
} continue;
}
/* SYS_FOREIGN.ID is defined as CHAR in the InnoDBinternalSQLparser,whichtranslates intotheincorrectVARCHAR(0).InnoDBdoes notenforcemaximumlengthsofcolumns,so thatiswhyanydatacanbeinsertedinthe firstplace.
Likewise,SYS_FOREIGN.FOR_NAME, SYS_FOREIGN.REF_NAME,SYS_FOREIGN_COLS.ID,are
defined as CHAR, and also they are part of a key. */
if (comp) { /* Add the extra size for ROW_FORMAT=COMPACT. ForROW_FORMAT=REDUNDANT,thesebyteswere
added to rec_max_size before this loop. */
rec_max_size += field_ext_max_size;
}
MY_ATTRIBUTE((nonnull,warn_unused_result)) /** Acquire a latch on the previous page without violating the latching order. @paramrw_latchthelatchonblock(RW_S_LATCHorRW_X_LATCH) @parampage_idpageidentifierwithvalidspaceidentifier @paramerrerrorcode @parammtrmini-transaction @retval0ifanerroroccurred @retval1ifthepagecouldbelatchedinthewrongorder
@retval -1 if the latch on block was temporarily released */ staticint btr_latch_prev(rw_lock_type_t rw_latch,
page_id_t page_id, dberr_t *err, mtr_t *mtr) noexcept
{
ut_ad(rw_latch == RW_S_LATCH || rw_latch == RW_X_LATCH);
Ifthereisaconflict,wewilltemporarilyreleaseourlatchonthe currentblockwhilewaitingforalatchontheleftsibling.The
buffer-fixes on both blocks will prevent eviction. */
retry: int ret= 1;
buf_block_t *prev=
buf_pool.page_fix(page_id, err, mtr->trx, buf_pool_t::FIX_NOWAIT); if (UNIV_UNLIKELY(!prev)) return0; if (prev == reinterpret_cast<buf_block_t*>(-1))
{ /* The block existed in buf_pool.page_hash, but not in a state that is safetoaccesswithoutwaitingforsomependingoperation,suchas buf_page_t::read_complete()orbuf_pool_t::unzip().
Retrywhiletemporarilyreleasingthesuccessorblock->page.lock
(but retaining a buffer-fix so that the block cannot be evicted. */
if (rw_latch == RW_S_LATCH)
block->page.lock.s_unlock(); else
block->page.lock.x_unlock();
switch (latch_mode) { case BTR_MODIFY_TREE:
rw_latch= RW_X_LATCH;
node_ptr_max_size= btr_node_ptr_max_size(index()); if (latch_by_caller)
{
ut_ad(mtr->memo_contains_flagged(&index()->lock, MTR_MEMO_X_LOCK)); break;
} if (lock_intention == BTR_INTENTION_DELETE)
{
compress_limit= BTR_CUR_PAGE_COMPRESS_LIMIT(index()); if (os_aio_pending_reads_approx() &&
trx_sys.history_size_approx() > BTR_CUR_FINE_HISTORY_LENGTH)
{ /* Most delete-intended operations are due to the purge of history.
Prioritize them when the history list is growing huge. */
mtr_x_lock_index(index(), mtr); break;
}
}
mtr_sx_lock_index(index(), mtr); break; #ifdef UNIV_DEBUG case BTR_CONT_MODIFY_TREE:
ut_ad("invalid mode" == 0); break; #endif case BTR_MODIFY_ROOT_AND_LEAF:
rw_latch= RW_SX_LATCH; /* fall through */ default: if (!latch_by_caller)
mtr_s_lock_index(index(), mtr);
}
dberr_t err;
if (!index()->table->space)
{
corrupted:
ut_ad("corrupted" == 0); // FIXME: remove this
err= DB_CORRUPTION;
func_exit: if (UNIV_LIKELY_NULL(heap))
mem_heap_free(heap); return err;
}
if (height == ULINT_UNDEFINED)
{ /* We are in the B-tree index root page. */ #ifdef BTR_CUR_ADAPT
info->root_guess= block; #endif
reached_root:
height= page_level;
tree_height= height + 1;
if (!height)
{ /* The root page is also a leaf page.
We may have to reacquire the page latch in a different mode. */ switch (rw_latch) { case RW_S_LATCH: if (!(latch_mode & BTR_SEARCH_LEAF))
{
rw_latch= RW_X_LATCH;
ut_ad(rw_lock_type_t(latch_mode & ~12) == RW_X_LATCH);
mtr->lock_register(block_savepoint, MTR_MEMO_PAGE_X_FIX); if (!block->page.lock.s_x_upgrade_try())
{
block->page.lock.s_unlock();
block->page.lock.x_lock(); /* Dropping the index tree (and freeing the root page)
should be impossible while we hold index()->lock. */
ut_ad(!block->page.is_freed());
page_level= btr_page_get_level(block->page.frame); if (UNIV_UNLIKELY(page_level != 0))
{ /* btr_root_raise_and_insert() was executed meanwhile */
ut_ad(mtr->memo_contains_flagged(&index()->lock,
MTR_MEMO_S_LOCK));
block->page.lock.x_u_downgrade();
block->page.lock.u_s_downgrade();
rw_latch= RW_S_LATCH;
mtr->lock_register(block_savepoint, MTR_MEMO_PAGE_S_FIX); goto reached_root;
}
}
} if (rw_latch != RW_S_LATCH) break; if (!latch_by_caller) /* Release the tree s-latch */
mtr->rollback_to_savepoint(savepoint, savepoint + 1); goto reached_latched_leaf; case RW_SX_LATCH:
ut_ad(latch_mode == BTR_MODIFY_ROOT_AND_LEAF);
static_assert(int{BTR_MODIFY_ROOT_AND_LEAF} == int{RW_SX_LATCH}, "");
rw_latch= RW_X_LATCH;
mtr->lock_register(block_savepoint, MTR_MEMO_PAGE_X_FIX);
block->page.lock.u_x_upgrade(); break; case RW_X_LATCH: if (latch_mode == BTR_MODIFY_TREE) goto reached_index_root_and_leaf; break; case RW_NO_LATCH:
ut_ad(0);
} goto reached_root_and_leaf;
}
} elseif (UNIV_UNLIKELY(height != page_level)) goto corrupted; else switch (latch_mode) { case BTR_MODIFY_TREE: break; case BTR_MODIFY_ROOT_AND_LEAF:
ut_ad((mtr->at_savepoint(block_savepoint - 1)->page.id().page_no() ==
index()->page) == (tree_height <= height + 2)); if (tree_height <= height + 2) /* Retain the root page latch. */ break; /* fall through */ default:
ut_ad(block_savepoint > savepoint);
mtr->rollback_to_savepoint(block_savepoint - 1, block_savepoint);
block_savepoint--;
}
if (!height)
{ /* We reached the leaf level. */
ut_ad(block == mtr->at_savepoint(block_savepoint));
switch (latch_mode) { default: break; case BTR_MODIFY_TREE: if (btr_cur_need_opposite_intention(block->page, index()->is_clust(),
lock_intention,
node_ptr_max_size, compress_limit,
page_cur.rec)) /* If the rec is the first or last in the page for pessimistic deleteintention,itmightcausenode_ptrinsertfortheupper
level. We should change the intention and retry. */
need_opposite_intention: return pessimistic_search_leaf(tuple, mode, mtr);
/* If the first or the last record of the page or the same key valuetothefirstrecordorlastrecord,thenanotherpagemight bechosenwhenBTR_CONT_MODIFY_TREE.So,theparentpageshould notreleasedtoavoidingdeadlockwithblockingtheanothersearch
with the same key value. */ const rec_t *first=
page_rec_get_next_const(page_get_infimum_rec(block->page.frame));
ulint matched_fields;
if (UNIV_UNLIKELY(!first)) goto corrupted; if (page_cur.rec == first ||
page_rec_is_last(page_cur.rec, block->page.frame))
{
same_key_root:
detected_same_key_root= true; break;
}
/* Latch the previous page if the node pointer is the leftmost
of the current page. */ int ret= btr_latch_prev(rw_latch, page_id, &err, mtr); if (!ret) goto func_exit;
ut_ad(block_savepoint + 2 == mtr->get_savepoint()); if (ret < 0)
{
up_match= 0, low_match= 0, up_bytes= 0, low_bytes= 0; /* While our latch on the level-2 page prevents splits or mergesofthislevel-1block,otherthreadsmayhave modifieditduetosplittingormergingsomelevel-0(leaf)
pages underneath it. Thus, we must search again. */ if (page_cur_search_with_match(tuple, page_mode,
&up_match, &low_match,
&page_cur, nullptr)) goto corrupted;
offsets= rec_get_offsets(page_cur.rec, index(), offsets, 0,
ULINT_UNDEFINED, &heap);
page_id.set_page_no(btr_node_ptr_get_child_page_no(page_cur.rec,
offsets));
}
}
rw_latch= rw_lock_type_t(latch_mode & (RW_X_LATCH | RW_S_LATCH)); break; case BTR_MODIFY_LEAF: case BTR_SEARCH_LEAF:
rw_latch= rw_lock_type_t(latch_mode); if (!not_first_access)
buf_read_ahead_linear(page_id); break; case BTR_MODIFY_TREE:
ut_ad(rw_latch == RW_X_LATCH);
/** Mark a non-leaf page "least recently used", but avoid invoking
buf_page_t::set_accessed(), because we do not want linear read-ahead */ staticvoid btr_cur_nonleaf_make_young(buf_page_t *bpage)
{ if (UNIV_UNLIKELY(buf_page_peek_if_too_old(bpage)))
buf_page_make_young(bpage);
}
#ifdef BTR_CUR_HASH_ADAPT /* We do a dirty read of btr_search.enabled here. We will recheck in btr_search_build_page_hash_index()beforebuildingapagehash
index, while holding search latch. */ if (!btr_search.is_enabled(index())) ; elseif (tuple->info_bits & REC_INFO_MIN_REC_FLAG) /* This may be a search tuple for btr_pcur_t::restore_position(). */
ut_ad(tuple->is_metadata() ||
(tuple->is_metadata(tuple->info_bits ^ REC_STATUS_INSTANT))); elseif (index()->table->is_temporary()); elseif (!rec_is_metadata(page_cur.rec, *index()) &&
index()->search_info.hash_analysis_useful())
search_info_update(*mtr); #endif/* BTR_CUR_HASH_ADAPT */
err= DB_SUCCESS;
}
func_exit: if (UNIV_LIKELY_NULL(heap))
mem_heap_free(heap); return err;
}
if (page_cur_search_with_match(tuple, page_mode, &up_match, &low_match,
&page_cur, nullptr)) goto corrupted;
page_id_t page_id{block->page.id()};
offsets= rec_get_offsets(page_cur.rec, index(), offsets, 0, ULINT_UNDEFINED,
&heap); /* Go to the child node */
page_id.set_page_no(btr_node_ptr_get_child_page_no(page_cur.rec, offsets));
/********************************************************************//**
Searches an index tree and positions a tree cursor on a given non-leaf level.
NOTE: n_fields_cmp in tuple must be set so that it cannot be compared
to node pointer page number fields on the upper levels of the tree!
cursor->up_match and cursor->low_match both will have sensible values.
Cursor is left at the place where an insert of the
search tuple should be performed in the B-tree. InnoDB does an insert
immediately after the cursor. Thus, the cursor may end up on a user record, or on a page infimum record.
@param level the tree level of search
@param tuple data tuple; NOTE: n_fields_cmp in tuple must be set so that
it cannot get compared to the node ptr page number field!
@param latch RW_S_LATCH or RW_X_LATCH
@param cursor tree cursor; the cursor page is s- or x-latched, but see also
above!
@param mtr mini-transaction
@return DB_SUCCESS on success or error code otherwise */
dberr_t btr_cur_search_to_nth_level(ulint level, const dtuple_t *tuple,
rw_lock_type_t rw_latch,
btr_cur_t *cursor, mtr_t *mtr)
{
dict_index_t *const index= cursor->index();
if (height == ULINT_UNDEFINED)
{ /* We are in the root node */
height= page_level; if (!height) goto corrupted;
cursor->tree_height= height + 1;
} elseif (height != ulint{page_level}) goto corrupted;
cursor->page_cur.block= block;
/* Search for complete index fields. */ if (page_cur_search_with_match(tuple, PAGE_CUR_LE, &cursor->up_match,
&cursor->low_match, &cursor->page_cur,
nullptr)) goto corrupted;
/* If this is the desired level, leave the loop */ if (level == height) goto func_exit;
ut_ad(height > level);
height--;
offsets = rec_get_offsets(cursor->page_cur.rec, index, offsets, 0,
ULINT_UNDEFINED, &heap); /* Go to the child node */
page_id.set_page_no(btr_node_ptr_get_child_page_no(cursor->page_cur.rec,
offsets));
block= nullptr; goto search_loop;
}
if (latch_mode == BTR_MODIFY_TREE)
{
node_ptr_max_size= btr_node_ptr_max_size(index); /* Most of delete-intended operations are purging. Free blocks andreadIObandwidthshouldbeprioritizedforthem,whenthe
history list is growing huge. */
savepoint++; if (lock_intention == BTR_INTENTION_DELETE)
{
compress_limit= BTR_CUR_PAGE_COMPRESS_LIMIT(index);
if (os_aio_pending_reads_approx() &&
trx_sys.history_size_approx() > BTR_CUR_FINE_HISTORY_LENGTH)
{
mtr_x_lock_index(index, mtr); goto index_locked;
}
}
mtr_sx_lock_index(index, mtr);
} else
{
static_assert(int{BTR_CONT_MODIFY_TREE} == (12 | BTR_MODIFY_LEAF), "");
ut_ad(!(latch_mode & 8)); /* This function doesn't need to lock left page of the leaf page */
static_assert(int{BTR_SEARCH_PREV} == (4 | BTR_SEARCH_LEAF), "");
latch_mode= btr_latch_mode(latch_mode & (RW_S_LATCH | RW_X_LATCH));
ut_ad(!latch_by_caller ||
mtr->memo_contains_flagged(&index->lock,
MTR_MEMO_SX_LOCK | MTR_MEMO_S_LOCK));
upper_rw_latch= RW_S_LATCH; if (!latch_by_caller)
{
savepoint++;
mtr_s_lock_index(index, mtr);
}
}
if (height == ULINT_UNDEFINED)
{ /* We are in the root node */
height= l; if (height); elseif (upper_rw_latch != root_leaf_rw_latch)
{ /* We should retry to get the page, because the root page
is latched with different level as a leaf page. */
ut_ad(n_blocks == 0);
ut_ad(root_leaf_rw_latch != RW_NO_LATCH);
upper_rw_latch= root_leaf_rw_latch;
mtr->rollback_to_savepoint(savepoint);
height= ULINT_UNDEFINED; continue;
} else
{
reached_leaf: constauto leaf_savepoint= mtr->get_savepoint();
ut_ad(leaf_savepoint);
ut_ad(block == mtr->at_savepoint(leaf_savepoint - 1));
if (latch_mode == BTR_MODIFY_TREE)
{ /* x-latch also siblings from left to right */ if (page_has_prev(block->page.frame) &&
!btr_latch_prev(RW_X_LATCH, block->page.id(), &err, mtr)) break; if (page_has_next(block->page.frame) &&
!btr_block_get(*index, btr_page_get_next(block->page.frame),
RW_X_LATCH, mtr, &err)) break;
if (latch_mode != BTR_MODIFY_TREE)
{ if (!height && first && first_access)
buf_read_ahead_linear(page_id_t(block->page.id().space(), page));
} elseif (btr_cur_need_opposite_intention(block->page, index->is_clust(),
lock_intention,
node_ptr_max_size, compress_limit,
page_cur.rec))
{
need_opposite_intention: /* If the rec is the first or last in the page for pessimistic deleteintention,itmightcausenode_ptrinsertfortheupper
level. We should change the intention and retry. */
mtr->rollback_to_savepoint(savepoint);
mtr->index_lock_upgrade(); /* X-latch all pages from now on */
latch_mode= BTR_CONT_MODIFY_TREE;
page= index->page;
height= ULINT_UNDEFINED;
n_blocks= 0; continue;
} else
{ if (!btr_cur_will_modify_tree(index, block->page.frame,
lock_intention, page_cur.rec,
node_ptr_max_size,
index->table->space->zip_size(), mtr))
{
ut_ad(n_blocks); /* release buffer-fixes on pages that will not be modified
(except the root) */ if (n_blocks > 1)
{
mtr->rollback_to_savepoint(savepoint + 1, savepoint + n_blocks - 1);
n_blocks= 1;
}
}
}
/*************************************************************//**
Inserts a record if there is enough space, orif enough space can
be freed by reorganizing. Differs from btr_cur_optimistic_insert because
no heuristics is applied to whether it pays to use CPU time for
reorganizing the page ornot.
@return pointer to inserted record if succeed, else NULL */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
rec_t*
btr_cur_insert_if_possible( /*=======================*/
btr_cur_t* cursor, /*!< in: cursor on page after which to insert;
cursor stays valid */ const dtuple_t* tuple, /*!< in: tuple to insert; the size info need not
have been stored to tuple */
rec_offs** offsets,/*!< out: offsets on *rec */
mem_heap_t** heap, /*!< in/out: pointer to memory heap, or NULL */
ulint n_ext, /*!< in: number of externally stored columns */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
page_cur_t* page_cursor;
rec_t* rec;
/* If the record did not fit, reorganize. Forcompressedpages,page_cur_tuple_insert()
attempted this already. */ if (!rec && !page_cur_get_page_zip(page_cursor)
&& btr_page_reorganize(page_cursor, mtr) == DB_SUCCESS) {
rec = page_cur_tuple_insert(page_cursor, tuple, offsets, heap,
n_ext, mtr);
}
/*************************************************************//** For an insert, checks the locks and does the undo logging if desired.
@return DB_SUCCESS, DB_LOCK_WAIT, DB_FAIL, or error number */
UNIV_INLINE MY_ATTRIBUTE((warn_unused_result, nonnull(2,3,5,6)))
dberr_t
btr_cur_ins_lock_and_undo( /*======================*/
ulint flags, /*!< in: undo logging and locking flags: if notzero,theparametersindexandthr
should be specified */
btr_cur_t* cursor, /*!< in: cursor on page after which to insert */
dtuple_t* entry, /*!< in/out: entry to insert */
que_thr_t* thr, /*!< in: query thread or NULL */
mtr_t* mtr, /*!< in/out: mini-transaction */ bool* inherit)/*!< out: true if the inserted new record maybe shouldinheritLOCK_GAPtypelocksfromthe
successor record */
{ if (!(~flags | (BTR_NO_UNDO_LOG_FLAG | BTR_KEEP_SYS_FLAG))) { return DB_SUCCESS;
}
/* Check if we have to wait for a lock: enqueue an explicit lock
request if yes */
rec_t* rec = btr_cur_get_rec(cursor);
dict_index_t* index = cursor->index();
/* Check if there is predicate or GAP lock preventing the insertion */ if (!(flags & BTR_NO_LOCKING_FLAG)) { constunsigned type = index->type; if (UNIV_UNLIKELY(type & DICT_SPATIAL)) {
lock_prdt_t prdt;
rtr_mbr_t mbr;
rtr_get_mbr_from_tuple(entry, &mbr);
/* Use on stack MBR variable to test if a lock is needed.Ifso,thepredicate(MBR)willbeallocated
from lock heap in lock_prdt_insert_check_and_lock() */
lock_init_prdt_from_mbr(&prdt, &mbr, 0, nullptr);
if (id.page_no() != FIL_NULL && space->acquire())
buf_read_page_background(id, space, trx);
id.set_page_no(next);
if (next != FIL_NULL && space->acquire())
buf_read_page_background(id, space, trx);
}
/*************************************************************//**
Tries to perform an insert to a page in an index tree, next to cursor.
It is assumed that mtr holds an x-latch on the page. The operation does not succeed if there is too little space on the page. If there is just
one record on the page, the insert will always succeed; this is to
prevent trying to split a page with just one record.
@return DB_SUCCESS, DB_LOCK_WAIT, DB_FAIL, or error number */
dberr_t
btr_cur_optimistic_insert( /*======================*/
ulint flags, /*!< in: undo logging and locking flags: if not zero,theparametersindexandthrshouldbe
specified */
btr_cur_t* cursor, /*!< in: cursor on page after which to insert;
cursor stays valid */
rec_offs** offsets,/*!< out: offsets on *rec */
mem_heap_t** heap, /*!< in/out: pointer to memory heap */
dtuple_t* entry, /*!< in/out: entry to insert */
rec_t** rec, /*!< out: pointer to inserted record if
succeed */
big_rec_t** big_rec,/*!< out: big rec vector whose fields have to
be stored externally by the caller */
ulint n_ext, /*!< in: number of externally stored columns */
que_thr_t* thr, /*!< in/out: query thread; can be NULL if !(~flags &(BTR_NO_LOCKING_FLAG
| BTR_NO_UNDO_LOG_FLAG)) */
mtr_t* mtr) /*!< in/out: mini-transaction; ifthisfunctionreturnsDB_SUCCESSon aleafpageofasecondaryindexina compressedtablespace,thecallermust mtr_commit(mtr)beforelatching
any further pages */
{
big_rec_t* big_rec_vec = NULL;
dict_index_t* index;
page_cur_t* page_cursor;
buf_block_t* block;
page_t* page;
rec_t* dummy; bool leaf; bool reorg __attribute__((unused)); bool inherit = true;
ulint rec_size;
dberr_t err;
if (UNIV_UNLIKELY(entry->is_alter_metadata())) {
ut_ad(leaf); goto convert_big_rec;
}
/* Calculate the record size when entry is converted to a record */
rec_size = rec_get_converted_size(index, entry, n_ext);
if (page_zip_rec_needs_ext(rec_size, page_is_comp(page),
dtuple_get_n_fields(entry),
block->zip_size())) {
convert_big_rec: /* The record is so big that we have to store some fields
externally on separate database pages */
big_rec_vec = dtuple_convert_big_rec(index, 0, entry, &n_ext);
if (block->page.zip.data && leaf
&& (page_get_data_size(page) + rec_size
>= dict_index_zip_pad_optimal_page_size(index))) { /* If compression padding tells us that insertion will resultintoopackeduppagei.e.:whichislikelyto causecompressionfailurethendon'tdoanoptimistic
insertion. */
fail:
err = DB_FAIL;
/* prefetch siblings of the leaf for the pessimistic
operation, if the page is leaf. */ if (leaf) {
btr_cur_prefetch_siblings(block, index, mtr->trx);
}
fail_err:
if (big_rec_vec) {
dtuple_convert_back_big_rec(index, entry, big_rec_vec);
}
if (page_has_garbage(page)) { if (max_size < BTR_CUR_PAGE_REORGANIZE_LIMIT
&& n_recs > 1
&& page_get_max_insert_size(page, 1) < rec_size) {
goto fail;
}
}
/* If there have been many consecutive inserts to the clusteredindexleafpageofanuncompressedtable,checkif wehavetosplitthepagetoreserveenoughfreespacefor
future updates of records. */
/*************************************************************//**
Performs an insert on a page of an index tree. It is assumed that mtr
holds an x-latch on the tree and on the cursor page. If the insert is
made on the leaf level, to avoid deadlocks, mtr must also own x-latches
to brothers of page, if those brothers exist.
@return DB_SUCCESS or error number */
dberr_t
btr_cur_pessimistic_insert( /*=======================*/
ulint flags, /*!< in: undo logging and locking flags: if not zero,theparameterthrshouldbe specified;ifnoundologgingisspecified, thenthecallermusthavereservedenough freeextentsinthefilespacesothatthe
insertion will certainly succeed */
btr_cur_t* cursor, /*!< in: cursor after which to insert;
cursor stays valid */
rec_offs** offsets,/*!< out: offsets on *rec */
mem_heap_t** heap, /*!< in/out: pointer to memory heap
that can be emptied */
dtuple_t* entry, /*!< in/out: entry to insert */
rec_t** rec, /*!< out: pointer to inserted record if
succeed */
big_rec_t** big_rec,/*!< out: big rec vector whose fields have to
be stored externally by the caller */
ulint n_ext, /*!< in: number of externally stored columns */
que_thr_t* thr, /*!< in/out: query thread; can be NULL if !(~flags &(BTR_NO_LOCKING_FLAG
| BTR_NO_UNDO_LOG_FLAG)) */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
dict_index_t* index = cursor->index();
big_rec_t* big_rec_vec = NULL; bool inherit = false;
uint32_t n_reserved = 0;
if (page_zip_rec_needs_ext(rec_get_converted_size(index, entry, n_ext),
index->table->not_redundant(),
dtuple_get_n_fields(entry),
btr_cur_get_block(cursor)->zip_size())
|| UNIV_UNLIKELY(entry->is_alter_metadata()
&& !dfield_is_ext(
dtuple_get_nth_field(
entry,
index->first_user_field())))) { /* The record is so big that we have to store some fields
externally on separate database pages */
if (UNIV_LIKELY_NULL(big_rec_vec)) { /* This should never happen, but we handle
the situation in a robust manner. */
ut_ad(0);
dtuple_convert_back_big_rec(index, entry, big_rec_vec);
}
if (!(flags & BTR_NO_LOCKING_FLAG)) {
ut_ad(!index->table->is_temporary()); if (dict_index_is_spatial(index)) { /* Do nothing */
} else { /* The cursor might be moved to the other page andthemaxtrxidfieldshouldbeupdatedafter
the cursor was fixed. */ if (!dict_index_is_clust(index)) {
page_update_max_trx_id(
btr_cur_get_block(cursor),
btr_cur_get_page_zip(cursor),
thr_get_trx(thr)->id, mtr);
}
if (!page_rec_is_infimum(btr_cur_get_rec(cursor))
|| !page_has_prev(btr_cur_get_page(cursor))) { /* split and inserted need to call
lock_update_insert() always. */
inherit = true;
}
}
}
if (!dict_index_is_clust(index)) {
ut_ad(dict_index_is_online_ddl(index)
== !!(flags & BTR_CREATE_FLAG));
/* We do undo logging only when we update a clustered index
record */ return(lock_sec_rec_modify_check_and_lock(
flags, btr_cur_get_block(cursor), rec,
index, thr, mtr));
}
/* Check if we have to wait for a lock: enqueue an explicit lock
request if yes */
/* During IMPORT the trx id in the record can be in the future, if the.ibdfileisbeingimportedfromanotherinstance.DuringIMPORT
roll_ptr will be 0. */
ut_ad(roll_ptr == 0 ||
lock_check_trx_id_sanity(trx_read_trx_id(rec + offset),
rec, index, offsets));
if (UNIV_LIKELY(index->trx_id_offset))
{ const rec_t *prev= page_rec_get_prev_const(rec); if (UNIV_UNLIKELY(!prev || prev == rec)) return DB_CORRUPTION; elseif (page_rec_is_infimum(prev)); else for (src= prev + offset; d < DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN; d++) if (src[d] != sys[d]) break; if (d > 6 && memcmp(dest, sys, d))
{ /* We save space by replacing a single record
Thetotalsizeis:x+13versusx+4+15-d=x+19-dbytes. Tosavespace,wemusthaved>6,thatis,thecompleteDB_TRX_IDand
the first byte(s) of DB_ROLL_PTR must match the previous record. */
memcpy(dest, src, d);
mtr->memmove(*block, dest - block->page.frame, src - block->page.frame,
d);
dest+= d;
len-= d; /* DB_TRX_ID,DB_ROLL_PTR must be unique in each record when
DB_TRX_ID refers to an active transaction. */
ut_ad(len);
} else
d= 0;
}
if (UNIV_LIKELY(len)) /* extra safety, to avoid corrupting the log */
mtr->memcpy<mtr_t::MAYBE_NOP>(*block, dest, sys + d, len);
return DB_SUCCESS;
}
/*************************************************************//**
See if there is enough place in the page modification log to log
an update-in-place.
@retval falseif out of space
@retval trueif enough place */ bool
btr_cur_update_alloc_zip_func( /*==========================*/
page_zip_des_t* page_zip,/*!< in/out: compressed page */
page_cur_t* cursor, /*!< in/out: B-tree page cursor */ #ifdef UNIV_DEBUG
rec_offs* offsets,/*!< in/out: offsets of the cursor record */ #endif/* UNIV_DEBUG */
ulint length, /*!< in: size needed */ bool create, /*!< in: true=delete-and-insert,
false=update-in-place */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
dict_index_t* index = cursor->index;
/* Have a local copy of the variables as these can change
dynamically. */ const page_t* page = page_cur_get_page(cursor);
#ifdef UNIV_DEBUG if (rec_offs_comp(offsets)) { switch (rec_get_status(rec)) { case REC_STATUS_ORDINARY: break; case REC_STATUS_INSTANT:
ut_ad(index->is_instant()); break; case REC_STATUS_NODE_PTR: case REC_STATUS_INFIMUM: case REC_STATUS_SUPREMUM:
ut_ad("wrong record status in update" == 0);
}
} #endif/* UNIV_DEBUG */
/** Check if a ROW_FORMAT=COMPRESSED page can be updated in place @paramcurcursorpointingtoROW_FORMAT=COMPRESSEDpage @paramoffsetsrec_get_offsets(btr_cur_get_rec(cur)) @paramupdateindexfieldsbeingupdated @parammtrmini-transaction @returntherecordintheROW_FORMAT=COMPRESSEDpage
@retval nullptr if the page cannot be updated in place */
ATTRIBUTE_COLD static
rec_t *btr_cur_update_in_place_zip_check(btr_cur_t *cur, rec_offs *offsets, const upd_t& update, mtr_t *mtr)
{
dict_index_t *index= cur->index();
ut_ad(!index->table->is_temporary());
switch (update.n_fields) { case0: /* We are only changing the delete-mark flag. */ break; case1: if (!index->is_clust() ||
update.fields[0].field_no != index->db_roll_ptr()) goto check_for_overflow; /* We are only changing the delete-mark flag and DB_ROLL_PTR. */ break; case2: if (!index->is_clust() ||
update.fields[0].field_no != index->db_trx_id() ||
update.fields[1].field_no != index->db_roll_ptr()) goto check_for_overflow; /* We are only changing DB_TRX_ID, DB_ROLL_PTR, and the delete-mark. Theycanbeupdatedinplaceintheuncompressedpartofthe
ROW_FORMAT=COMPRESSED page. */ break;
check_for_overflow: default: if (!btr_cur_update_alloc_zip(btr_cur_get_page_zip(cur),
btr_cur_get_page_cur(cur),
offsets, rec_offs_size(offsets), false, mtr)) return nullptr;
}
return btr_cur_get_rec(cur);
}
/*************************************************************//**
Updates a record when the update causes no size changes in its fields.
We assume here that the ordering fields of the record donot change.
@return locking or undo log related error code, or
@retval DB_SUCCESS on success
@retval DB_ZIP_OVERFLOW if there is not enough space left
on a ROW_FORMAT=COMPRESSED page */
dberr_t
btr_cur_update_in_place( /*====================*/
ulint flags, /*!< in: undo logging and locking flags */
btr_cur_t* cursor, /*!< in: cursor on the record to update; cursorstaysvalidandpositionedonthe
same record */
rec_offs* offsets,/*!< in/out: offsets on cursor->page_cur.rec */ const upd_t* update, /*!< in: update vector */
ulint cmpl_info,/*!< in: compiler info on secondary index
updates */
que_thr_t* thr, /*!< in: query thread */
trx_id_t trx_id, /*!< in: transaction id */
mtr_t* mtr) /*!< in/out: mini-transaction; if this isasecondaryindex,thecallermust mtr_commit(mtr)beforelatchingany
further pages */
{
dict_index_t* index;
rec_t* rec;
roll_ptr_t roll_ptr = 0;
ulint was_delete_marked;
/* Check that enough space is available on the compressed page. */ if (UNIV_LIKELY_NULL(page_zip)
&& !(rec = btr_cur_update_in_place_zip_check(
cursor, offsets, *update, mtr))) { return DB_ZIP_OVERFLOW;
}
/* Do lock checking and undo logging */ if (dberr_t err = btr_cur_upd_lock_and_undo(flags, cursor, offsets,
update, cmpl_info,
thr, mtr, &roll_ptr)) { return err;
}
was_delete_marked = rec_get_deleted_flag(
rec, page_is_comp(buf_block_get_frame(block))); /* In delete-marked records, DB_TRX_ID must always refer to an
existing undo log record. */
ut_ad(!was_delete_marked
|| !dict_index_is_clust(index)
|| row_get_rec_trx_id(rec, index, offsets));
#ifdef BTR_CUR_HASH_ADAPT
{ auto part = block->index
? &btr_search.get_part(*index) : nullptr; if (part) { /* TO DO: Can we skip this if none of the fields index->search_info->curr_n_fields
are being updated? */
/* The function row_upd_changes_ord_field_binary
does not work on a secondary index. */
if (!dict_index_is_clust(index)
|| row_upd_changes_ord_field_binary(
index, update, thr, NULL, NULL)) {
ut_ad(!(update->info_bits
& REC_INFO_MIN_REC_FLAG)); /* Remove possible hash index pointer
to this record */
btr_search_update_hash_on_delete(cursor);
}
if (was_delete_marked
&& !rec_get_deleted_flag(
rec, page_is_comp(buf_block_get_frame(block)))) { /* The new updated record owns its possible externally
stored fields */
/** Trim a metadata record during the rollback of instant ALTER TABLE. @param[in]entrymetadatatuple @param[in]indexprimarykey @param[in]updateupdatevectorfortherollback
@param[in,out] trx transaction */
ATTRIBUTE_COLD staticvoid btr_cur_trim_alter_metadata(dtuple_t* entry, const dict_index_t* index, const upd_t* update, trx_t *trx)
{
ut_ad(index->is_instant());
ut_ad(update->is_alter_metadata());
ut_ad(entry->is_alter_metadata());
/** Trim an update tuple due to instant ADD COLUMN, if needed. Fornormalrecords,thetrailinginstantlyaddedfieldsthatmatch theinitialdefaultvaluesareomitted.
@param[in,out]entryindexentry @param[in]indexindex @param[in]updateupdatevector
@param[in] thr execution thread */ staticinline void
btr_cur_trim(
dtuple_t* entry, const dict_index_t* index, const upd_t* update, const que_thr_t* thr)
{ if (!index->is_instant()) {
} elseif (UNIV_UNLIKELY(update->is_metadata())) { /* We are either updating a metadata record (instantALTERTABLEonatablewhereinstantALTERwas
already executed) or rolling back such an operation. */
ut_ad(!upd_get_nth_field(update, 0)->orig_len);
ut_ad(entry->is_metadata());
trx_t* const trx{thr->graph->trx};
if (trx->in_rollback) { /* This rollback can occur either as part of ha_innobase::commit_inplace_alter_table()rolling backafterafailedinnobase_add_instant_try(), oraspartofcrashrecovery.Eitherway,the tablewillbeinthedatadictionarycache,with theinstantlyaddedcolumnsgoingtoberemoved
later in the rollback. */
ut_ad(index->table->cached); /* The DB_TRX_ID,DB_ROLL_PTR are always last, andthereshouldbesomechangetorollback. Thefirstfieldintheupdatevectoristhe firstinstantlyaddedcolumnloggedby
innobase_add_instant_try(). */
ut_ad(update->n_fields > 2); if (update->is_alter_metadata()) {
btr_cur_trim_alter_metadata(
entry, index, update, trx); return;
}
ut_ad(!entry->is_alter_metadata());
/*************************************************************//**
Tries to update a record on a page in an index tree. It is assumed that mtr
holds an x-latch on the page. The operation does not succeed if there is too
little space on the page orif the update would result in too empty a page,
so that tree compression is recommended. We assume here that the ordering
fields of the record donot change.
@return error code, including
@retval DB_SUCCESS on success
@retval DB_OVERFLOW if the updated record does not fit
@retval DB_UNDERFLOW if the page would become too empty
@retval DB_ZIP_OVERFLOW if there is not enough space left
on a ROW_FORMAT=COMPRESSED page */
dberr_t
btr_cur_optimistic_update( /*======================*/
ulint flags, /*!< in: undo logging and locking flags */
btr_cur_t* cursor, /*!< in: cursor on the record to update; cursorstaysvalidandpositionedonthe
same record */
rec_offs** offsets,/*!< out: offsets on cursor->page_cur.rec */
mem_heap_t** heap, /*!< in/out: pointer to NULL or memory heap */ const upd_t* update, /*!< in: update vector; this must also
contain trx id and roll ptr fields */
ulint cmpl_info,/*!< in: compiler info on secondary index
updates */
que_thr_t* thr, /*!< in: query thread */
trx_id_t trx_id, /*!< in: transaction id */
mtr_t* mtr) /*!< in/out: mini-transaction; if this isasecondaryindex,thecallermust mtr_commit(mtr)beforelatchingany
further pages */
{
dict_index_t* index;
page_cur_t* page_cursor;
dberr_t err;
buf_block_t* block;
page_t* page;
page_zip_des_t* page_zip;
rec_t* rec;
ulint max_size;
ulint new_rec_size;
ulint old_rec_size;
dtuple_t* new_entry;
roll_ptr_t roll_ptr;
ulint i;
*offsets = rec_get_offsets(rec, index, *offsets, index->n_core_fields,
ULINT_UNDEFINED, heap); #ifdefined UNIV_DEBUG || defined UNIV_BLOB_LIGHT_DEBUG /* Blob pointer can be null if InnoDB was killed or
ran out of space while allocating a page. */
ut_a(!rec_offs_any_null_extern(rec, *offsets)
|| thr_get_trx(thr)->in_rollback); #endif/* UNIV_DEBUG || UNIV_BLOB_LIGHT_DEBUG */
if (UNIV_LIKELY(!update->is_metadata())
&& !row_upd_changes_field_size_or_external(index, *offsets,
update)) {
/* The simplest and the most common case: the update does not changethesizeofanyfieldandnoneoftheupdatedfieldsis externallystoredinrecorupdate,andthereisenoughspace
on the compressed page to log the update. */
/* The page containing the clustered index record correspondingtonew_entryislatchedinmtr.
Thus the following call is safe. */
row_upd_index_replace_new_col_vals_index_pos(new_entry, index, update,
*heap);
btr_cur_trim(new_entry, index, update, thr);
old_rec_size = rec_offs_size(*offsets);
new_rec_size = rec_get_converted_size(index, new_entry, 0);
/* We limit max record size to 16k even for 64k page size. */ if (new_rec_size >= COMPRESSED_REC_MAX_DATA_SIZE ||
(!dict_table_is_comp(index->table)
&& new_rec_size >= REDUNDANT_REC_MAX_DATA_SIZE)) {
err = DB_OVERFLOW; goto func_exit;
}
if (UNIV_UNLIKELY(page_get_data_size(page)
- old_rec_size + new_rec_size
< BTR_CUR_PAGE_COMPRESS_LIMIT(index))) { /* The page would become too empty. When innodb_index_shrink isOFF,onlytreatthisasDB_UNDERFLOWiftherecordis actuallyshrinking;otherwiseafreshlysplitpagewouldbe
re-merged even though the update is growing the record. */ if (new_rec_size < old_rec_size
|| btr_cur_index_shrink) {
err = DB_UNDERFLOW; goto func_exit;
}
}
/* We do not attempt to reorganize if the page is compressed.
This is because the page may fail to compress after reorganization. */
max_size = page_zip
? page_get_max_insert_size(page, 1)
: (old_rec_size
+ page_get_max_insert_size_after_reorganize(page, 1));
if (!(((max_size >= BTR_CUR_PAGE_REORGANIZE_LIMIT)
&& (max_size >= new_rec_size))
|| (page_get_n_recs(page) <= 1))) { /* There was not enough space, or it did not pay to reorganize:forsimplicity,wedecidewhattodoassuminga
reorganization is needed, though it might not be necessary */
err = DB_OVERFLOW; goto func_exit;
}
/* Do lock checking and undo logging */
err = btr_cur_upd_lock_and_undo(flags, cursor, *offsets,
update, cmpl_info,
thr, mtr, &roll_ptr); if (err != DB_SUCCESS) { goto func_exit;
}
/* Ok, we may do the replacement. Store on the page infimum the explicitlocksonrec,beforedeletingrec(seethecommentin
btr_cur_pessimistic_update). */ if (index->has_locking()) {
lock_rec_store_on_page_infimum(block, rec);
}
if (UNIV_UNLIKELY(update->is_metadata())) {
ut_ad(new_entry->is_metadata());
ut_ad(index->is_instant()); /* This can be innobase_add_instant_try() performing a subsequentinstantADDCOLUMN,oritsrollbackby
row_undo_mod_clust_low(). */
ut_ad(flags & BTR_NO_LOCKING_FLAG);
} else {
btr_search_update_hash_on_delete(cursor);
}
page_cur_delete_rec(page_cursor, *offsets, mtr);
if (!page_cur_move_to_prev(page_cursor)) { return DB_CORRUPTION;
}
if (!(flags & BTR_KEEP_SYS_FLAG)) {
btr_cur_write_sys(new_entry, index, trx_id, roll_ptr);
}
if (UNIV_UNLIKELY(update->is_metadata())) { /* We must empty the PAGE_FREE list, because if this wasarollback,theshortenedmetadatarecord wouldhavetoomanyfields,andwewouldbeunableto
know the size of the freed record. */
err = btr_page_reorganize(page_cursor, mtr); if (err != DB_SUCCESS) { goto func_exit;
}
} else { /* Restore the old explicit lock state on the record */
lock_rec_restore_from_page_infimum(*block, rec,
block->page.id());
}
ut_ad(err == DB_SUCCESS); if (!page_cur_move_to_next(page_cursor)) {
corrupted: return DB_CORRUPTION;
}
if (err != DB_SUCCESS) {
func_exit: /* prefetch siblings of the leaf for the pessimistic
operation. */
btr_cur_prefetch_siblings(block, index, mtr->trx);
}
return(err);
}
/*************************************************************//** If, in a split, a new supremum record was created as the predecessor of the
updated record, the supremum record must inherit exactly the locks on the
updated record. In the split it may have inherited locks from the successor
of the updated record, which is not correct. This function restores the
right locks for the new supremum. */ static
dberr_t
btr_cur_pess_upd_restore_supremum( /*==============================*/
buf_block_t* block, /*!< in: buffer block of rec */ const rec_t* rec, /*!< in: updated record */
mtr_t* mtr) /*!< in: mtr */
{
page_t* page;
page = buf_block_get_frame(block);
if (page_rec_get_next(page_get_infimum_rec(page)) != rec) { /* Updated record is not the first user record on its page */ return DB_SUCCESS;
}
/*************************************************************//**
Performs an update of a record on a page of a tree. It is assumed
that mtr holds an x-latch on the tree and on the cursor page. If the
update is made on the leaf level, to avoid deadlocks, mtr must also
own x-latches to brothers of page, if those brothers exist. We assume
here that the ordering fields of the record donot change.
@return DB_SUCCESS or error code */
dberr_t
btr_cur_pessimistic_update( /*=======================*/
ulint flags, /*!< in: undo logging, locking, and rollback
flags */
btr_cur_t* cursor, /*!< in/out: cursor on the record to update; cursormaybecomeinvalidif*big_rec==NULL
|| !(flags & BTR_KEEP_POS_FLAG) */
rec_offs** offsets,/*!< out: offsets on cursor->page_cur.rec */
mem_heap_t** offsets_heap, /*!< in/out: pointer to memory heap
that can be emptied */
mem_heap_t* entry_heap, /*!< in/out: memory heap for allocating
big_rec and the index tuple */
big_rec_t** big_rec,/*!< out: big rec vector whose fields have to
be stored externally by the caller */
upd_t* update, /*!< in/out: update vector; this is allowed to alsocontaintrxidandrollptrfields. Non-updatedcolumnsthataremovedoffpagewill
be appended to this. */
ulint cmpl_info,/*!< in: compiler info on secondary index
updates */
que_thr_t* thr, /*!< in: query thread */
trx_id_t trx_id, /*!< in: transaction id */
mtr_t* mtr) /*!< in/out: mini-transaction; must be
committed before latching any further pages */
{
big_rec_t* big_rec_vec = NULL;
big_rec_t* dummy_big_rec;
dict_index_t* index;
buf_block_t* block;
rec_t* rec;
page_cur_t* page_cursor;
dberr_t err;
dberr_t optim_err;
roll_ptr_t roll_ptr; bool was_first;
uint32_t n_reserved = 0;
*offsets = NULL;
*big_rec = NULL;
block = btr_cur_get_block(cursor);
index = cursor->index();
/* The page containing the clustered index record correspondingtonew_entryislatchedinmtr.Ifthe clusteredindexrecordisdelete-marked,thenitsexternally storedfieldscannothavebeenpurgedyet,becausethenthe purgewouldalsohaveremovedtheclusteredindexrecord
itself. Thus the following call is safe. */
row_upd_index_replace_new_col_vals_index_pos(new_entry, index, update,
entry_heap);
btr_cur_trim(new_entry, index, update, thr);
/* We have to set appropriate extern storage bits in the new
record to be inserted: we have to remember which fields were such */
if ((flags & BTR_NO_UNDO_LOG_FLAG)
&& rec_offs_any_extern(*offsets)) { /* We are in a transaction rollback undoing a row update:wemustfreepossibleexternallystoredfields whichgotnewvaluesintheupdate,iftheyarenot inheritedvalues.Theycanbeinheritedifwehave updatedtheprimarykeytoanothervalue,andthen
update it back again. */
/* Do lock checking and undo logging */
err = btr_cur_upd_lock_and_undo(flags, cursor, *offsets,
update, cmpl_info,
thr, mtr, &roll_ptr); if (err != DB_SUCCESS) { goto err_exit;
}
if (optim_err == DB_OVERFLOW) { /* First reserve enough free space for the file segments oftheindextree,sothattheupdatewillnotfailbecause
of lack of space */
if (!(flags & BTR_KEEP_SYS_FLAG)) {
btr_cur_write_sys(new_entry, index, trx_id, roll_ptr);
}
if (UNIV_UNLIKELY(is_metadata)) {
ut_ad(new_entry->is_metadata());
ut_ad(index->is_instant()); /* This can be innobase_add_instant_try() performing a subsequentinstantALTERTABLE,oritsrollbackby
row_undo_mod_clust_low(). */
ut_ad(flags & BTR_NO_LOCKING_FLAG);
} else {
btr_search_update_hash_on_delete(cursor);
/* Store state of explicit locks on rec on the page infimumrecord,beforedeletingrec.Thepageinfimum actsasadummycarrierofthelocks,takingcarealso oflockreleases,beforewecanmovethelocksbackon theactualrecord.Thereisaspecialcase:ifweare insertingontherootpageandtheinsertcausesa callofbtr_root_raise_and_insert.Thereforewecannot inthelocksystemdeletethelockstructssetonthe rootpageeveniftherootpagecarriesjustnode
pointers. */
lock_rec_store_on_page_infimum(block, rec);
}
if (!page_cur_move_to_prev(page_cursor)) {
err = DB_CORRUPTION; goto return_after_reservations;
}
/* Force a page split instead of the in-place reinsert below when (cheapestchecksfirst;themostexpensiveonelast): -optimisticupdatereturnedDB_OVERFLOW(thefitproblemthis optimizationaddresses;DB_UNDERFLOWmeansthepageistoosparse andthelegacycompresspathistherightanswer); -innodb_index_shrinkisOFF(theopt-in); -uncompressedpage:thereorganize-fitcheckusesuncompressed-page accountingandisnotmeaningfulforROW_FORMAT=COMPRESSED; -pagestillhasatleastonerecordafterthedeleteabove(so theforcedsplitwouldproduceatleast1+1,not0+1); -thepageistoofulltosatisfyBTR_CUR_PAGE_REORGANIZE_LIMIT afterareorganize,sothelegacyin-placeretrywouldjustchurn; -thenewrecordisstrictlylargerthantheoldone(only size-growingUPDATEsshouldpaythecostofanearlysplit).This rec_get_converted_size()walkseveryfield,soitischeckedlast.
rec=NULL falls through to the split path. */ if (optim_err == DB_OVERFLOW
&& !btr_cur_index_shrink
&& !buf_block_get_page_zip(block)
&& page_get_n_recs(block->page.frame) > 0
&& page_get_max_insert_size_after_reorganize(
block->page.frame, 1) < BTR_CUR_PAGE_REORGANIZE_LIMIT
&& rec_get_converted_size(index, new_entry, n_ext)
> rec_offs_size(*offsets)) {
rec = NULL;
} else {
rec = btr_cur_insert_if_possible(cursor, new_entry,
offsets, offsets_heap, n_ext, mtr);
}
if (rec) {
page_cursor->rec = rec;
if (UNIV_UNLIKELY(is_metadata)) { /* We must empty the PAGE_FREE list, because if this wasarollback,theshortenedmetadatarecord wouldhavetoomanyfields,andwewouldbeunableto
know the size of the freed record. */
err = btr_page_reorganize(page_cursor, mtr); if (err != DB_SUCCESS) { goto return_after_reservations;
}
rec = page_cursor->rec;
rec_offs_make_valid(rec, index, true, *offsets); if (page_cursor->block->page.id().page_no()
== index->page) {
btr_set_instant(page_cursor->block, *index,
mtr);
}
} else {
lock_rec_restore_from_page_infimum(
*btr_cur_get_block(cursor), rec,
block->page.id());
}
if (!rec_get_deleted_flag(rec, rec_offs_comp(*offsets))
|| rec_is_alter_metadata(rec, *index)) { /* The new inserted record owns its possible externally
stored fields */
btr_cur_unmark_extern_fields(btr_cur_get_block(cursor),
rec, index, *offsets, mtr);
} else { /* In delete-marked records, DB_TRX_ID must
always refer to an existing undo log record. */
ut_ad(row_get_rec_trx_id(rec, index, *offsets));
}
if (btr_cur_compress_if_useful(cursor, adjust, mtr)) { if (adjust) {
rec_offs_make_valid(page_cursor->rec, index, true, *offsets);
}
}
#if0// FIXME: this used to be a no-op, and will cause trouble if enabled if (!big_rec_vec
&& page_is_leaf(block->page.frame)
&& !dict_index_is_online_ddl(index)) {
mtr->release(index->lock); /* NOTE: We cannot release root block latch here, because it
has segment header and already modified in most of cases.*/
} #endif
err = DB_SUCCESS; goto return_after_reservations;
} else { /* If the page is compressed and it initially compressesverywell,andthereisasubsequentinsert ofabadly-compressingrecord,itispossiblefor btr_cur_optimistic_update()toreturnDB_UNDERFLOWand
btr_cur_insert_if_possible() to return FALSE. */
ut_ad(page_zip || optim_err != DB_UNDERFLOW);
}
if (big_rec_vec != NULL) {
ut_ad(page_is_leaf(block->page.frame));
ut_ad(dict_index_is_clust(index));
ut_ad(flags & BTR_KEEP_POS_FLAG);
/* btr_page_split_and_insert() in btr_cur_pessimistic_insert()invokes mtr->release(index->lock). Wemustkeeptheindex->lockwhenwecreateda big_rec,sothatrow_upd_clust_rec()canstorethe
big_rec in the same mini-transaction. */
/* Was the record to be updated positioned as the first user
record on its page? */
was_first = page_cur_is_before_first(page_cursor);
/* Lock checks and undo logging were already performed by btr_cur_upd_lock_and_undo().Wedonottry btr_cur_optimistic_insert()because
btr_cur_insert_if_possible() already failed above. */
err = btr_cur_pessimistic_insert(BTR_NO_UNDO_LOG_FLAG
| BTR_NO_LOCKING_FLAG
| BTR_KEEP_SYS_FLAG,
cursor, offsets, offsets_heap,
new_entry, &rec,
&dummy_big_rec, n_ext, NULL, mtr); if (err) { /* This should happen when InnoDB tries to extend the
tablespace */
ut_ad(err == DB_OUT_OF_FILE_SPACE); return err;
}
ut_a(rec);
ut_a(dummy_big_rec == NULL);
ut_ad(rec_offs_validate(rec, cursor->index(), *offsets));
page_cursor->rec = rec;
/* Multiple transactions cannot simultaneously operate on the sametemp-tableinparallel. max_trx_idisignoredfortemptablesbecauseitnotrequired
for MVCC. */ if (!index->is_primary() && !index->table->is_temporary()) { /* Update PAGE_MAX_TRX_ID in the index page header. Itwasnotupdatedbybtr_cur_pessimistic_insert()
because of BTR_NO_LOCKING_FLAG. */
page_update_max_trx_id(btr_cur_get_block(cursor),
btr_cur_get_page_zip(cursor),
trx_id, mtr);
}
if (!rec_get_deleted_flag(rec, rec_offs_comp(*offsets))) { /* The new inserted record owns its possible externally
stored fields */ #ifdef UNIV_ZIP_DEBUG
ut_a(!page_zip
|| page_zip_validate(page_zip, block->page.frame, index)); #endif/* UNIV_ZIP_DEBUG */
btr_cur_unmark_extern_fields(btr_cur_get_block(cursor), rec,
index, *offsets, mtr);
} else { /* In delete-marked records, DB_TRX_ID must
always refer to an existing undo log record. */
ut_ad(row_get_rec_trx_id(rec, index, *offsets));
}
if (UNIV_UNLIKELY(is_metadata)) { /* We must empty the PAGE_FREE list, because if this wasarollback,theshortenedmetadatarecord wouldhavetoomanyfields,andwewouldbeunableto
know the size of the freed record. */
err = btr_page_reorganize(page_cursor, mtr); if (err != DB_SUCCESS) { goto return_after_reservations;
}
rec = page_cursor->rec;
} else {
lock_rec_restore_from_page_infimum(
*btr_cur_get_block(cursor), rec, block->page.id());
}
/* If necessary, restore also the correct lock state for a new, precedingsupremumrecordcreatedinapagesplit.Whiletheold recordwasnonexistent,thesupremummighthaveinheriteditslocks
from a wrong record. */
if (!was_first) {
err = btr_cur_pess_upd_restore_supremum(
btr_cur_get_block(cursor), rec, mtr);
}
/***********************************************************//**
Marks a clustered index record deleted. Writes an undo log record to
undo log on thisdelete marking. Writes in the trx id field the id
of the deleting transaction, and in the roll ptr field pointer to the
undo log record created.
@return DB_SUCCESS, DB_LOCK_WAIT, or error number */
dberr_t
btr_cur_del_mark_set_clust_rec( /*===========================*/
buf_block_t* block, /*!< in/out: buffer block of the record */
rec_t* rec, /*!< in/out: record */
dict_index_t* index, /*!< in: clustered index of the record */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec) */
que_thr_t* thr, /*!< in: query thread */ const dtuple_t* entry, /*!< in: dtuple for the deleting record, also
contains the virtual cols if there are any */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
roll_ptr_t roll_ptr;
dberr_t err;
trx_t* trx;
if (rec_get_deleted_flag(rec, rec_offs_comp(offsets))) { /* We may already have delete-marked this record
when executing an ON DELETE CASCADE operation. */
ut_ad(row_get_rec_trx_id(rec, index, offsets)
== thr_get_trx(thr)->id); return(DB_SUCCESS);
}
/*==================== B-TREE RECORD REMOVE =========================*/
/*************************************************************//**
Tries to compress a page of the tree if it seems useful. It is assumed
that mtr holds an x-latch on the tree and on the cursor page. To avoid
deadlocks, mtr must also own x-latches to brothers of page, if those
brothers exist. NOTE: it is assumed that the caller has reserved enough
free extents so that the compression will always succeed if done!
@return whether compression occurred */ bool
btr_cur_compress_if_useful( /*=======================*/
btr_cur_t* cursor, /*!< in/out: cursor on the page to compress; cursordoesnotstayvalidif!adjustand
compression occurs */ bool adjust, /*!< in: whether the cursor position should be
adjusted even when compression occurs */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
ut_ad(mtr->memo_contains_flagged(&cursor->index()->lock,
MTR_MEMO_X_LOCK | MTR_MEMO_SX_LOCK));
ut_ad(mtr->memo_contains_flagged(btr_cur_get_block(cursor),
MTR_MEMO_PAGE_X_FIX));
/*******************************************************//**
Removes the record on which the tree cursor is positioned on a leaf page.
It is assumed that the mtr has an x-latch on the page where the cursor is
positioned, but no latch on the whole tree.
@return error code
@retval DB_FAIL if the page would become too empty */
dberr_t
btr_cur_optimistic_delete( /*======================*/
btr_cur_t* cursor, /*!< in: cursor on leaf page, on the record to delete;cursorstaysvalid:ifdeletion succeeds,onfunctionexititpointstothe
successor of the deleted record */
ulint flags, /*!< in: BTR_CREATE_FLAG or 0 */
mtr_t* mtr) /*!< in: mtr; if this function returns TRUEonaleafpageofasecondary index,themtrmustbecommitted
before latching any further pages */
{
buf_block_t* block;
rec_t* rec;
mem_heap_t* heap = NULL;
rec_offs offsets_[REC_OFFS_NORMAL_SIZE];
rec_offs* offsets = offsets_;
rec_offs_init(offsets_);
{ if (UNIV_UNLIKELY(rec_get_info_bits(rec, page_is_comp(
block->page.frame))
& REC_INFO_MIN_REC_FLAG)) { /* This should be rolling back instant ADD COLUMN. Ifthisisarecoveredtransaction,then index->is_instant()willholduntilthe
insert into SYS_COLUMNS is rolled back. */
ut_ad(cursor->index()->table->supports_instant());
ut_ad(cursor->index()->is_primary());
ut_ad(!buf_block_get_page_zip(block));
page_cur_delete_rec(btr_cur_get_page_cur(cursor),
offsets, mtr); /* We must empty the PAGE_FREE list, because afterrollback,thisdeletedmetadatarecord wouldhavetoomanyfields,andwewouldbe
unable to know the size of the freed record. */
err = btr_page_reorganize(btr_cur_get_page_cur(cursor),
mtr); goto func_exit;
} else { if (!flags) {
lock_update_delete(block, rec);
}
func_exit: if (UNIV_LIKELY_NULL(heap)) {
mem_heap_free(heap);
}
return err;
}
/*************************************************************//**
Removes the record on which the tree cursor is positioned. Tries
to compress the page if its fillfactor drops below a threshold orif it is the only page on the level. It is assumed that mtr holds
an x-latch on the tree and on the cursor page. To avoid deadlocks,
mtr must also own x-latches to brothers of page, if those brothers
exist.
@returnTRUEif compression occurred andFALSEifnotor something
wrong. */
ibool
btr_cur_pessimistic_delete( /*=======================*/
dberr_t* err, /*!< out: DB_SUCCESS or DB_OUT_OF_FILE_SPACE; thelattermayoccurbecausewemayhave toupdatenodepointersonupperlevels, andinthecaseofvariablelengthkeys
these may actually grow in size */
ibool has_reserved_extents, /*!< in: TRUE if the callerhasalreadyreservedenoughfree extentssothatheknowsthattheoperation
will succeed */
btr_cur_t* cursor, /*!< in: cursor on the record to delete; ifcompressiondoesnotoccur,thecursor staysvalid:itpointstosuccessorof
deleted record on function exit */
ulint flags, /*!< in: BTR_CREATE_FLAG or 0 */ bool rollback,/*!< in: performing rollback? */
mtr_t* mtr) /*!< in: mtr */
{
buf_block_t* block;
page_t* page;
page_zip_des_t* page_zip;
dict_index_t* index;
rec_t* rec;
uint32_t n_reserved = 0;
ibool ret = FALSE;
mem_heap_t* heap;
rec_offs* offsets; #ifdef UNIV_DEBUG bool parent_latched = false; #endif/* UNIV_DEBUG */
block = btr_cur_get_block(cursor);
page = buf_block_get_frame(block);
index = btr_cur_get_index(cursor);
if (!has_reserved_extents) { /* First reserve enough free space for the file segments oftheindextree,sothatthenodepointerupdateswill
not fail because of lack of space */
if (page_is_leaf(page)) { constbool is_metadata = rec_is_metadata(
rec, page_is_comp(block->page.frame)); if (UNIV_UNLIKELY(is_metadata)) { /* This should be rolling back instant ALTER TABLE. Ifthisisarecoveredtransaction,then index->is_instant()willholduntilthe
insert into SYS_COLUMNS is rolled back. */
ut_ad(rollback);
ut_ad(index->table->supports_instant());
ut_ad(index->is_primary());
} elseif (flags == 0) {
lock_update_delete(block, rec);
}
if (block->page.id().page_no() != index->page) { if (page_get_n_recs(page) < 2) { goto discard_page;
}
} elseif (page_get_n_recs(page) == 1
+ (index->is_instant() && !is_metadata)
&& !index->must_avoid_clear_instant_add()) { /* The whole index (and table) becomes logically empty. Emptythewholepage.Thatis,ifwearedeletingthe onlyuserrecord,alsodeletethemetadatarecord ifoneexistsforinstantADDCOLUMN (notgenericALTERTABLE). Ifwearedeletingthemetadatarecord (intherollbackofinstantALTERTABLE)andthe
table becomes empty, clean up the whole page. */
page_cur_set_after_last(
block,
btr_cur_get_page_cur(cursor));
ret = TRUE; goto return_after_reservations;
}
}
if (UNIV_LIKELY(!is_metadata)) {
btr_search_update_hash_on_delete(cursor);
} else {
page_cur_delete_rec(btr_cur_get_page_cur(cursor),
offsets, mtr); /* We must empty the PAGE_FREE list, because afterrollback,thisdeletedmetadatarecord wouldcarrytoomanyfields,andwewouldbe
unable to know the size of the freed record. */
*err = btr_page_reorganize(btr_cur_get_page_cur(cursor),
mtr);
ut_ad(!ret); goto err_exit;
}
} elseif (UNIV_UNLIKELY(page_rec_is_first(rec, page))) { if (page_rec_is_last(rec, page)) {
discard_page:
ut_ad(page_get_n_recs(page) == 1); /* If there is only one record, drop
the whole page. */
if (!page_has_prev(page)) { /* If we delete the leftmost node pointer on a non-leaflevel,wemustmarkthenewleftmostnode
pointer as the predefined minimum record */
min_mark_next_rec = true;
} elseif (index->is_spatial()) { /* For rtree, if delete the leftmost node pointer,
we need to update parent page. */
rtr_mbr_t father_mbr;
rec_t* father_rec;
rec_offs* offsets;
ulint len;
/* SPATIAL INDEX never use U locks; we can allow page merges whileholdingXlockonthespatialindextree. Donotallowmergesofnon-leafB-treepagesunlessitis
safe to do so. */
{ constbool allow_merge = page_is_leaf(page)
|| dict_index_is_spatial(index)
|| btr_cur_will_modify_tree(
index, page, BTR_INTENTION_DELETE, rec,
btr_node_ptr_max_size(index),
block->zip_size(), mtr);
page_cur_delete_rec(btr_cur_get_page_cur(cursor),
offsets, mtr);
if (min_mark_next_rec) {
btr_set_min_rec_mark(next_rec, *block, mtr);
}
#if0// FIXME: this used to be a no-op, and will cause trouble if enabled if (page_is_leaf(page)
&& !dict_index_is_online_ddl(index)) {
mtr->release(index->lock); /* NOTE: We cannot release root block latch here, because it
has segment header and already modified in most of cases.*/
} #endif
/** Represents the cursor for the number of rows estimation. The contentisusedforlevel-by-leveldivingandestimationthenumberofrows
on each level. */ class btr_est_cur_t
{ /* Assume a page like: records:(inf,a,b,c,d,sup) indexoftherecord:0,1,2,3,4,5
*/
/** Index of the record where the page cursor stopped on this level (indexinalphabeticalorder).Intheaboveexample,ifthesearchstoppedon
record 'c', then nth_rec will be 3. */
ulint m_nth_rec;
/** Number of the records on the page, not counting inf and sup.
In the above example n_recs will be 4. */
ulint m_n_recs;
/** Search tuple */ const dtuple_t &m_tuple; /** Cursor search mode */
page_cur_mode_t m_mode; /** Page cursor which is used for search */
page_cur_t m_page_cur; /** Page id of the page to get on level down, can differ from m_block->page.idatthemomentwhenthechild'spageidisalreadyfound,but
the child's block has not fetched yet */
page_id_t m_page_id; /** Current block */
buf_block_t *m_block; /** Page search mode, can differ from m_mode for non-leaf pages, see c-tor
comments for details */
page_cur_mode_t m_page_mode;
/** Matched fields and bytes which are used for on-page search, see
btr_cur_t::(up|low)_(match|bytes) comments for details */
uint16_t m_up_match= 0;
uint16_t m_up_bytes= 0;
uint16_t m_low_match= 0;
uint16_t m_low_bytes= 0;
/** Does search on the current page. If there is no border in m_tuple, then justmovethecursortothemostleftorrightrecord. @paramlevelcurrentlevelontree. @paramroot_heightrootheight @paramlefttrueifthisisleftborder,falseotherwise.
@return true on success, false otherwise. */ bool search_on_page(ulint level, ulint root_height, bool left)
{ if (level != btr_page_get_level(m_block->page.frame)) returnfalse;
/** Read page id of the current record child. @paramoffsetsoffsetsarray.
@param heap heap for offsets array */ void read_child_page_id(rec_offs **offsets, mem_heap_t **heap)
{ const rec_t *node_ptr= page_cur_get_rec(&m_page_cur);
/* FIXME: get the child page number directly without computing offsets */
*offsets= rec_get_offsets(node_ptr, index(), *offsets, 0, ULINT_UNDEFINED,
heap);
/* Go to the child node */
m_page_id.set_page_no(btr_node_ptr_get_child_page_no(node_ptr, *offsets));
}
/** @return true if left border should be counted */ bool should_count_the_left_border() const
{ if (dtuple_get_n_fields(&m_tuple) > 0)
{
ut_ad(!page_rec_is_infimum(page_cur_get_rec(&m_page_cur))); return !page_rec_is_supremum(page_cur_get_rec(&m_page_cur));
}
ut_ad(page_rec_is_infimum(page_cur_get_rec(&m_page_cur))); returnfalse;
}
/** @return true if right border should be counted */ bool should_count_the_right_border() const
{ if (dtuple_get_n_fields(&m_tuple) > 0)
{ const rec_t *rec= page_cur_get_rec(&m_page_cur);
ut_ad(!(m_mode == PAGE_CUR_L && page_rec_is_supremum(rec)));
return (m_mode == PAGE_CUR_LE /* if the range is '<=' */ /* and the record was found */
&& m_low_match >= dtuple_get_n_fields(&m_tuple)) ||
(m_mode == PAGE_CUR_L /* or if the range is '<' */ /* and there are any records to match the criteria, i.e. if the minimumrecordonthetreeis5andx<7isspecifiedthenthe cursorwillbepositionedat5andweshouldcounttheborder, butifx<2isspecified,thenthecursorwillbepositionedat
'inf' and we should not count the border */
&& !page_rec_is_infimum(rec)); /* Notice that for "WHERE col <= 'foo'" the server passes to ha_innobase::records_in_range():min_key=NULL(left-unbounded)whichis expectedmax_key='foo'flag=HA_READ_AFTER_KEY(PAGE_CUR_G),whichis unexpected-onewouldexpectflag=HA_READ_KEY_OR_PREV(PAGE_CUR_LE).In thiscasethecursorwillbepositionedonthefirstrecordtotheright oftherequestedone(canalsobepositionedonthe'sup')andweshould
not count the right border. */
}
ut_ad(page_rec_is_supremum(page_cur_get_rec(&m_page_cur)));
/* The range specified is without a right border, just 'x > 123' or'x>=123'andsearch_on_page()positionedthecursoronthe
supremum record on the rightmost page, which must not be counted. */ returnfalse;
}
/** @return current page id */
page_id_t page_id() const { return m_page_id; }
/** Copies block pointer and savepoint from another btr_est_cur_t in the case ifbothleftandrightbordercursorspointtothesameblock.
@param o reference to the other btr_est_cur_t object. */ void set_block(const btr_est_cur_t &o) { m_block= o.m_block; }
/** @return current record number. */
ulint nth_rec() const { return m_nth_rec; }
/** @return number of records in the current page. */
ulint n_recs() const { return m_n_recs; }
};
/** Estimate the number of rows between the left record of the path and the rightone(non-inclusive)forthecertainlevelonaB-tree.Thisfunction startsfromthepagenexttotheleftpageandreadsafewpagestotheright, countingtheirrecords.Ifwereachtherightpagequicklythenweknowexactly howmanyrecordstherearebetweenleftandrightrecordsandweset is_n_rows_exacttotrue.Aftersomepageislatched,thepreviouspageis unlatched.Ifwecannotreachtherightpagequicklythenwecalculatethe averagenumberofrecordsinthepagesscannedsofarandassumethatallpages thatwedidnotscanuptotherightpagecontainthesamenumberofrecords, thenwemultiplythataveragetothenumberofpagesbetweenrightandleft records(whichisn_rows_on_prev_level).Inthiscasewesetis_n_rows_exactto false. @paramlevelcurrentlevel. @paramleft_curthecursoroftheleftpage. @paramright_page_norightpagenumber. @paramn_rows_on_prev_levelnumberofrowsonthepreviouslevel. @param[out]is_n_rows_exacttrueifexactrowsnumberisreturned. @param[in,out]mtrmtr,
@return number of rows, not including the borders (exact or estimated). */ static ha_rows btr_estimate_n_rows_in_range_on_level(
ulint level, btr_est_cur_t &left_cur, uint32_t right_page_no,
ha_rows n_rows_on_prev_level, bool &is_n_rows_exact, mtr_t &mtr)
{
ha_rows n_rows= 0;
uint n_pages_read= 0; /* Do not read more than this number of pages in order not to hurt performancewiththiscodewhichisjustanestimation.Ifwereadthismany pagesbeforereachingright_page_no,thenweestimatetheaveragefromthe
pages scanned so far. */ static constexpr uint n_pages_read_limit= 9;
buf_block_t *block= nullptr; const dict_index_t *index= left_cur.index();
/* Assume by default that we will scan all pages between left and right(non
inclusive) pages */
is_n_rows_exact= true;
/* Add records from the left page which are to the right of the record which servesasaleftborderoftherange,ifany(wedon'tincludetherecord
itself in this count). */ if (left_cur.nth_rec() <= left_cur.n_recs())
{
n_rows+= left_cur.n_recs() - left_cur.nth_rec();
}
/* Count the records in the pages between left and right (non inclusive)
pages */
if (prev_block)
{
ulint savepoint = mtr.get_savepoint(); /* Index s-lock, p1, p2 latches, can also be p1 and p2 parent latch if
they are not diverged */
ut_ad(savepoint >= 3);
mtr.rollback_to_savepoint(savepoint - 2, savepoint - 1);
}
if (!block || btr_page_get_level(buf_block_get_frame(block)) != level) goto inexact;
page= buf_block_get_frame(block);
/* It is possible but highly unlikely that the page was originally written byanoldversionofInnoDBthatdidnotinitializeFIL_PAGE_TYPEonother thanB-treepages.Forexample,thiscouldbeanalmost-emptyBLOBpage thathappenstocontainthemagicvaluesinthefields
that we checked above. */
n_pages_read++;
n_rows+= page_get_n_recs(page);
page_id.set_page_no(btr_page_get_next(page));
if (n_pages_read == n_pages_read_limit)
{ /* We read too many pages or we reached the end of the level
without passing through right_page_no. */ goto inexact;
}
if (n_pages_read > 0)
{ /* The number of pages on this level is n_rows_on_prev_level,multiplyitbythe
average number of recs per page so far */
n_rows= n_rows_on_prev_level * n_rows / n_pages_read;
} else
{
n_rows= 10;
}
return (n_rows);
}
/** Estimates the number of rows in a given index range. Do search in the leftpage,theniftherearepagesbetweenleftandrightones,readafew pagestotheright,iftherightpageisreached,fetchitandcounttheexact numberofrows,otherwisecounttheestimated(see btr_estimate_n_rows_in_range_on_level()fordetails)numberifrows,and fetchtherightpage.Ifleavesarereached,unlatchnon-leafpagesexcept therightleafparent.Aftertherightleafpageisfetched,commitmtr. @paramtrxtransaction @paramindexB-tree @paramrange_startfirstkey @paramrange_endlastkey
@return estimated number of rows; */
ha_rows btr_estimate_n_rows_in_range(trx_t *trx, dict_index_t *index,
btr_pos_t *range_start,
btr_pos_t *range_end)
{
DBUG_ENTER("btr_estimate_n_rows_in_range");
if (UNIV_UNLIKELY(index->page == FIL_NULL || index->is_corrupted()))
DBUG_RETURN(0);
/* This becomes true when the two paths do not pass through the same pages
anymore. */ bool diverged= false; /* This is the height, i.e. the number of levels from the root, where paths
are not the same or adjacent any more. */
ulint divergence_height= ULINT_UNDEFINED; bool should_count_the_left_border= true; bool should_count_the_right_border= true; bool is_n_rows_exact= true;
ha_rows n_rows= 0;
/* Loop and search until we arrive at the desired level. */
search_loop: if (!p1.fetch_child(height, mtr, p2.block())) goto error;
if (height == ULINT_UNDEFINED)
{ /* We are in the root node */
height= btr_page_get_level(buf_block_get_frame(p1.block()));
root_height= height;
}
if (!height)
{
p1.set_page_mode_for_leaves();
p2.set_page_mode_for_leaves();
}
if (p1.page_id() == p2.page_id())
p2.set_block(p1); else
{
ut_ad(diverged); if (divergence_height != ULINT_UNDEFINED) { /* We need to call p1.search_on_page() here as btr_estimate_n_rows_in_range_on_level()usesp1.m_n_recsand
p1.m_nth_rec. */ if (!p1.search_on_page(height, root_height, true)) goto error;
n_rows= btr_estimate_n_rows_in_range_on_level(
height, p1, p2.page_id().page_no(), n_rows, is_n_rows_exact, mtr);
} if (!p2.fetch_child(height, mtr, nullptr)) goto error;
}
if (height == 0) /* There is no need to release non-leaf pages here as they must already be unlatchedinbtr_est_cur_t::fetch_child().Trytosearchonpagesafter
releasing the index latch, to decrease contention. */
mtr.rollback_to_savepoint(0, 1);
/* There is no need to search on left page if divergence_height!=ULINT_UNDEFINED,asitwasalreadysearchedbefore
btr_estimate_n_rows_in_range_on_level() call */ if (divergence_height == ULINT_UNDEFINED &&
!p1.search_on_page(height, root_height, true)) goto error;
if (!p2.search_on_page(height, root_height, false)) goto error;
if (!diverged && (p1.nth_rec() != p2.nth_rec()))
{
ut_ad(p1.page_id() == p2.page_id());
diverged= true; if (p1.nth_rec() < p2.nth_rec())
{ /* We do not count the borders (nor the left nor the right one), thus
"- 1". */
n_rows= p2.nth_rec() - p1.nth_rec() - 1;
if (n_rows > 0)
{ /* There is at least one row between the two borders pointed to by p1 andp2,soonthelevelbelowtheslotswillpointtonon-adjacent
pages. */
divergence_height= root_height - height;
}
} else
{ /* It is possible that p1->nth_rec > p2->nth_rec if, for example, we have asinglepagetreewhichcontains(inf,5,6,supr)andweselectwherex >20andx<30;inthiscasep1->nth_recwillpointtothesuprrecord
and p2->nth_rec will point to 6. */
n_rows= 0;
should_count_the_left_border= false;
should_count_the_right_border= false;
}
} elseif (diverged && divergence_height == ULINT_UNDEFINED)
{
if (p1.nth_rec() < p1.n_recs())
{
n_rows+= p1.n_recs() - p1.nth_rec();
}
if (p2.nth_rec() > 1)
{
n_rows+= p2.nth_rec() - 1;
}
}
} elseif (divergence_height != ULINT_UNDEFINED)
{ /* All records before the right page was already counted. Add records from p2->page_nowhicharetotheleftoftherecordwhichserversasaright borderoftherange,ifany(wedon'tincludetherecorditselfinthis
count). */ if (p2.nth_rec() > 1)
n_rows+= p2.nth_rec() - 1;
}
/* Here none of the borders were counted. For example, if on the leaf level wedescendedto: (inf,a,b,c,d,e,f,sup) ^^ path1path2
then n_rows will be 2 (c and d). */
if (is_n_rows_exact)
{ /* Only fiddle to adjust this off-by-one if the number is exact, otherwise
we do much grosser adjustments below. */
/* If both paths end up on the same record on the leaf level. */ if (p1.page_id() == p2.page_id() && p1.nth_rec() == p2.nth_rec())
{
/* n_rows can be > 0 here if the paths were first different and then convergedtothesamerecordontheleaflevel. Forexample: SELECT...LIKE'wait/synch/rwlock%' mode1=PAGE_CUR_GE, tuple1="wait/synch/rwlock" path1[0]={nth_rec=58,n_recs=58, page_no=3,page_level=1} path1[1]={nth_rec=56,n_recs=55, page_no=119,page_level=0}
/* If the range is such that we should count both borders, then avoid countingthatrecordtwice-onceasaleftborderandonceasaright
border. Some of the borders should not be counted, e.g. [3,3). */
n_rows= should_count_the_left_border && should_count_the_right_border;
} else
n_rows+= should_count_the_left_border + should_count_the_right_border;
}
if (root_height > divergence_height && !is_n_rows_exact) /* In trees whose height is > 1 our algorithm tends to underestimate:
multiply the estimate by 2: */
n_rows*= 2;
#ifdef NOT_USED /* Do not estimate the number of rows in the range to over 1 / 2 of the
estimated rows in the whole table */
if (n_rows > table_n_rows / 2 && !is_n_rows_exact)
{
n_rows= table_n_rows / 2;
/* If there are just 0 or 1 rows in the table, then we estimate all rows
are in the range */
if (n_rows == 0)
n_rows= table_n_rows;
} #else if (n_rows > table_n_rows)
n_rows= table_n_rows; #endif
DBUG_RETURN(n_rows);
error:
mtr.commit(); if (UNIV_LIKELY_NULL(heap))
mem_heap_free(heap);
DBUG_RETURN(0);
}
/*================== EXTERNAL STORAGE OF BIG FIELDS ===================*/
/***********************************************************//**
Gets the offset of the pointer to the externally stored part of a field.
@return offset of the pointer to the externally stored part */ static
ulint
btr_rec_get_field_ref_offs( /*=======================*/ const rec_offs* offsets,/*!< in: array returned by rec_get_offsets() */
ulint n) /*!< in: index of the external field */
{
ulint field_ref_offs;
ulint local_len;
/** Gets a pointer to the externally stored part of a field. @paramrecrecord @paramoffsetsrec_get_offsets(rec) @paramnindexoftheexternallystoredfield
@return pointer to the externally stored part */ #define btr_rec_get_field_ref(rec, offsets, n) \
((rec) + btr_rec_get_field_ref_offs(offsets, n))
/** Gets the externally stored size of a record, in units of a database page. @param[in]recrecord @param[in]offsetsarrayreturnedbyrec_get_offsets()
@return externally stored part, in units of a database page */
ulint
btr_rec_get_externally_stored_len( const rec_t* rec, const rec_offs* offsets)
{
ulint n_fields;
ulint total_extern_len = 0;
ulint i;
/*******************************************************************//**
Sets the ownership bit of an externally stored field in a record. */ static void
btr_cur_set_ownership_of_extern_field( /*==================================*/
buf_block_t* block, /*!< in/out: index page */
rec_t* rec, /*!< in/out: clustered index record */
dict_index_t* index, /*!< in: index of the page */ const rec_offs* offsets,/*!< in: array returned by rec_get_offsets() */
ulint i, /*!< in: field number */ bool val, /*!< in: value to set */
mtr_t* mtr) /*!< in: mtr, or NULL if not logged */
{
byte* data;
ulint local_len;
ulint byte_val;
data = rec_get_nth_field(rec, offsets, i, &local_len);
ut_ad(rec_offs_nth_extern(offsets, i));
ut_a(local_len >= BTR_EXTERN_FIELD_REF_SIZE);
if (UNIV_LIKELY_NULL(block->page.zip.data)) {
mach_write_to_1(data + local_len + BTR_EXTERN_LEN, byte_val);
page_zip_write_blob_ptr(block, rec, index, offsets, i, mtr);
} else {
mtr->write<1,mtr_t::MAYBE_NOP>(*block, data + local_len
+ BTR_EXTERN_LEN, byte_val);
}
}
/*******************************************************************//**
Marks non-updated off-page fields as disowned by this record. The ownership
must be transferred to the updated record which is inserted elsewhere in the
index tree. In purge only the owner of externally stored field is allowed
to free the field. */ void
btr_cur_disown_inherited_fields( /*============================*/
buf_block_t* block, /*!< in/out: index page */
rec_t* rec, /*!< in/out: record in a clustered index */
dict_index_t* index, /*!< in: index of the page */ const rec_offs* offsets,/*!< in: array returned by rec_get_offsets() */ const upd_t* update, /*!< in: update vector */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
ut_ad(rec_offs_validate(rec, index, offsets));
ut_ad(!rec_offs_comp(offsets) || !rec_get_node_ptr_flag(rec));
ut_ad(rec_offs_any_extern(offsets));
for (uint16_t i = 0; i < rec_offs_n_fields(offsets); i++) { if (rec_offs_nth_extern(offsets, i)
&& !upd_get_field_by_field_no(update, i, false)) {
btr_cur_set_ownership_of_extern_field(
block, rec, index, offsets, i, false, mtr);
}
}
}
/*******************************************************************//**
Marks all extern fields in a record as owned by the record. This function
should be called if the delete mark of a record is removed: a notdelete
marked record always owns all its extern fields. */ static void
btr_cur_unmark_extern_fields( /*=========================*/
buf_block_t* block, /*!< in/out: index page */
rec_t* rec, /*!< in/out: record in a clustered index */
dict_index_t* index, /*!< in: index of the page */ const rec_offs* offsets,/*!< in: array returned by rec_get_offsets() */
mtr_t* mtr) /*!< in: mtr, or NULL if not logged */
{
ut_ad(!rec_offs_comp(offsets) || !rec_get_node_ptr_flag(rec)); if (!rec_offs_any_extern(offsets)) { return;
}
const ulint n = rec_offs_n_fields(offsets);
for (ulint i = 0; i < n; i++) { if (rec_offs_nth_extern(offsets, i)) {
btr_cur_set_ownership_of_extern_field(
block, rec, index, offsets, i, true, mtr);
}
}
}
/*******************************************************************//**
Returns the length of a BLOB part stored on the header page.
@return part length */ static
uint32_t
btr_blob_get_part_len( /*==================*/ const byte* blob_header) /*!< in: blob header */
{ return(mach_read_from_4(blob_header + BTR_BLOB_HDR_PART_LEN));
}
/*******************************************************************//**
Returns the page number where the next BLOB part is stored.
@return page number or FIL_NULL if no more pages */ static
uint32_t
btr_blob_get_next_page_no( /*======================*/ const byte* blob_header) /*!< in: blob header */
{ return(mach_read_from_4(blob_header + BTR_BLOB_HDR_NEXT_PAGE_NO));
}
/** Deallocate a buffer block that was reserved for a BLOB part. @paramblockbufferblock @paramallflagwhethertoremoveaROW_FORMAT=COMPRESSEDpage
@param mtr mini-transaction to commit */ staticvoid btr_blob_free(buf_block_t *block, bool all, mtr_t *mtr)
{
ut_ad(mtr->memo_contains_flagged(block, MTR_MEMO_PAGE_X_FIX));
block->page.fix(); #ifdef UNIV_DEBUG const page_id_t page_id{block->page.id()};
buf_pool_t::hash_chain &chain= buf_pool.page_hash.cell_get(page_id.fold()); #endif
mtr->commit();
if (!buf_LRU_free_page(&block->page, all) && all && block->page.zip.data) /* Attempt to deallocate the redundant copy of the uncompressed page
if the whole ROW_FORMAT=COMPRESSED block cannot be deallocated. */
buf_LRU_free_page(&block->page, false);
mysql_mutex_unlock(&buf_pool.mutex);
}
/** Helper class used while writing blob pages, during insert or update. */ struct btr_blob_log_check_t { /** Persistent cursor on a clusterex index record with blobs. */
btr_pcur_t* m_pcur; /** Mini transaction holding the latches for m_pcur */
mtr_t* m_mtr; /** rec_get_offsets(rec, index); offset of clust_rec */ const rec_offs* m_offsets; /** The block containing clustered record */
buf_block_t** m_block; /** The clustered record pointer */
rec_t** m_rec; /** The blob operation code */ enum blob_op m_op;
/** Check if there is enough space in log file. Commit and re-start the
mini transaction. */ void check()
{
dict_index_t* index = m_pcur->index();
ulint offs = 0;
uint32_t page_no = FIL_NULL;
if (UNIV_UNLIKELY(page_no != FIL_NULL)) {
dberr_t err; if (UNIV_LIKELY(index->page != page_no)) {
ut_a(btr_root_block_get(index, RW_SX_LATCH,
m_mtr, &err));
}
m_pcur->btr_cur.page_cur.block = btr_block_get(
*index, page_no, RW_X_LATCH, m_mtr); /* The page should not be evicted or corrupted while
we are holding a buffer-fix on it. */
m_pcur->btr_cur.page_cur.block->page.unfix();
m_pcur->btr_cur.page_cur.rec
= m_pcur->btr_cur.page_cur.block->page.frame
+ offs;
} else {
ut_ad(m_pcur->rel_pos == BTR_PCUR_ON);
mtr_sx_lock_index(index, m_mtr);
ut_a(m_pcur->restore_position(
BTR_MODIFY_ROOT_AND_LEAF_ALREADY_LATCHED,
m_mtr) == btr_pcur_t::SAME_ALL);
}
/*******************************************************************//**
Stores the fields in big_rec_vec to the tablespace and puts pointers to
them in rec. The extern flags in rec will have to be set beforehand.
The fields are stored on pages allocated from leaf node
file segment of the index tree.
TODO: If the allocation extends the tablespace, it will not be redo logged, in
any mini-transaction. Tablespace extension should be redo-logged, so that
recovery will not fail when the big_rec was written to the extended portion of
the file, in case the file was somehow truncated in the crash.
@return DB_SUCCESS or DB_OUT_OF_FILE_SPACE */
dberr_t
btr_store_big_rec_extern_fields( /*============================*/
btr_pcur_t* pcur, /*!< in: a persistent cursor */
rec_offs* offsets, /*!< in/out: rec_get_offsets() on pcur.the"externalstorage"flags inoffsetswillcorrectlycorrespond
to rec when this function returns */ const big_rec_t*big_rec_vec, /*!< in: vector containing fields
to be stored externally */
mtr_t* btr_mtr, /*!< in/out: mtr containing the latchestotheclusteredindex.canbe
committed and restarted. */ enum blob_op op) /*! in: operation code */
{
byte* field_ref;
ulint extern_len;
ulint store_len;
ulint i;
mtr_t mtr{btr_mtr->trx};
mem_heap_t* heap = NULL;
page_zip_des_t* page_zip;
z_stream c_stream;
dberr_t error = DB_SUCCESS;
dict_index_t* index = pcur->index();
buf_block_t* rec_block = btr_pcur_get_block(pcur);
rec_t* rec = btr_pcur_get_rec(pcur);
#ifdefined UNIV_DEBUG || defined UNIV_BLOB_LIGHT_DEBUG /* All pointers to externally stored columns in the record musteitherbezeroortheymustbepointerstoinherited
columns, owned by this record or an earlier record version. */ for (i = 0; i < big_rec_vec->n_fields; i++) {
field_ref = btr_rec_get_field_ref(
rec, offsets, big_rec_vec->fields[i].field_no);
ut_a(!(field_ref[BTR_EXTERN_LEN] & BTR_EXTERN_OWNER_FLAG)); /* Either this must be an update in place, ortheBLOBmustbeinherited,ortheBLOBpointer
must be zero (will be written in this function). */
ut_a(op == BTR_STORE_UPDATE
|| (field_ref[BTR_EXTERN_LEN] & BTR_EXTERN_INHERITED_FLAG)
|| !memcmp(field_ref, field_ref_zero,
BTR_EXTERN_FIELD_REF_SIZE));
} #endif/* UNIV_DEBUG || UNIV_BLOB_LIGHT_DEBUG */
/* Space available in compressed page to carry blob data */ const ulint payload_size_zip = rec_block->physical_size()
- FIL_PAGE_DATA;
/* Space available in uncompressed page to carry blob data */ const ulint payload_size = payload_size_zip
- (BTR_BLOB_HDR_SIZE + FIL_PAGE_DATA_END);
/* We have to create a file segment to the tablespace
for each field and put the pointer to the field in rec */
for (i = 0; i < big_rec_vec->n_fields; i++) { const ulint field_no = big_rec_vec->fields[i].field_no;
field_ref = btr_rec_get_field_ref(rec, offsets, field_no); #ifdefined UNIV_DEBUG || defined UNIV_BLOB_LIGHT_DEBUG /* A zero BLOB pointer should have been initially inserted. */
ut_a(!memcmp(field_ref, field_ref_zero,
BTR_EXTERN_FIELD_REF_SIZE)); #endif/* UNIV_DEBUG || UNIV_BLOB_LIGHT_DEBUG */
extern_len = big_rec_vec->fields[i].len;
MEM_CHECK_DEFINED(big_rec_vec->fields[i].data, extern_len);
ut_a(extern_len > 0);
uint32_t prev_page_no = FIL_NULL;
if (page_zip) { int err = deflateReset(&c_stream);
ut_a(err == Z_OK);
func_exit: if (page_zip) {
deflateEnd(&c_stream);
}
if (heap != NULL) {
mem_heap_free(heap);
}
#ifdefined UNIV_DEBUG || defined UNIV_BLOB_LIGHT_DEBUG /* All pointers to externally stored columns in the record
must be valid. */ for (i = 0; i < rec_offs_n_fields(offsets); i++) { if (!rec_offs_nth_extern(offsets, i)) { continue;
}
/* The pointer must not be zero if the operation
succeeded. */
ut_a(0 != memcmp(field_ref, field_ref_zero,
BTR_EXTERN_FIELD_REF_SIZE)
|| error != DB_SUCCESS); /* The column must not be disowned by this record. */
ut_a(!(field_ref[BTR_EXTERN_LEN] & BTR_EXTERN_OWNER_FLAG));
} #endif/* UNIV_DEBUG || UNIV_BLOB_LIGHT_DEBUG */ return(error);
}
/** Check the FIL_PAGE_TYPE on an uncompressed BLOB page. @paramblockuncompressedBLOBpage @paramopoperation
@return whether the type is invalid */ staticbool btr_check_blob_fil_page_type(const buf_block_t& block, constchar *op)
{
uint16_t type= fil_page_get_type(block.page.frame);
if (UNIV_LIKELY(type == FIL_PAGE_TYPE_BLOB)); elseif (fil_space_t *space= fil_space_t::get(block.page.id().space()))
{ /* Old versions of InnoDB did not initialize FIL_PAGE_TYPE on BLOB pages.Donotprintanythingaboutthetypemismatchwhenreading
a BLOB page that may be from old versions. */ bool fail= space->full_crc32() || DICT_TF_HAS_ATOMIC_BLOBS(space->flags); if (fail)
sql_print_error("InnoDB: FIL_PAGE_TYPE=%u on BLOB %s file %s page %u",
type, op, space->chain.start->name,
block.page.id().page_no());
space->release(); return fail;
} returnfalse;
}
/*******************************************************************//**
Frees the space in an externally stored field to the file space
management if the field in data is owned by the externally stored field,
in a rollback we may have the additional condition that the field must not be inherited. */ void
btr_free_externally_stored_field( /*=============================*/
dict_index_t* index, /*!< in: index of the data, the index treeMUSTbeX-latched;ifthetree heightis1,thenalsotherootpage mustbeX-latched!(thisisrelevant inthecasethisfunctioniscalled frompurgewhere'data'islocatedon anundologpage,notanindex
page) */
byte* field_ref, /*!< in/out: field reference */ const rec_t* rec, /*!< in: record containing field_ref, for
page_zip_write_blob_ptr(), or NULL */ const rec_offs* offsets, /*!< in: rec_get_offsets(rec, index),
or NULL */
buf_block_t* block, /*!< in/out: page of field_ref */
ulint i, /*!< in: field number of field_ref;
ignored if rec == NULL */ bool rollback, /*!< in: performing rollback? */
mtr_t* local_mtr) /*!< in: mtr containingthelatchtodataanan
X-latch to the index tree */
{ const uint32_t space_id = mach_read_from_4(
field_ref + BTR_EXTERN_SPACE_ID);
if (UNIV_UNLIKELY(!memcmp(field_ref, field_ref_zero,
BTR_EXTERN_FIELD_REF_SIZE))) { /* In the rollback, we may encounter a clustered index recordwithsomeunwrittenoff-pagecolumns.Thereis
nothing to free then. */
ut_a(rollback); return;
}
const ulint ext_zip_size = index->table->space->zip_size(); /* !rec holds in a call from purge when field_ref is in an undo page */
ut_ad(rec || !block->page.zip.data);
for (mtr_t mtr{local_mtr->trx};;) {
mtr.start();
mtr.set_spaces(*local_mtr);
mtr.set_log_mode_sub(*local_mtr);
if (/* There is no external storage data */
page_no == FIL_NULL /* This field does not own the externally stored field */
|| (mach_read_from_1(field_ref + BTR_EXTERN_LEN)
& BTR_EXTERN_OWNER_FLAG) /* Rollback and inherited field */
|| (rollback
&& (mach_read_from_1(field_ref + BTR_EXTERN_LEN)
& BTR_EXTERN_INHERITED_FLAG))) {
skip_free: /* Do not free */
mtr.commit();
/* The buffer pool block containing the BLOB pointer is exclusivelylatchedbylocal_mtr.Tosatisfysomedesign
constraints, we must recursively latch it in mtr as well. */
block->fix();
block->page.lock.x_lock();
if (ext_zip_size) { /* Note that page_zip will be NULL
in row_purge_upd_exist_or_extern(). */ switch (fil_page_get_type(page)) { case FIL_PAGE_TYPE_ZBLOB: case FIL_PAGE_TYPE_ZBLOB2: break; default:
MY_ASSERT_UNREACHABLE();
} const uint32_t next_page_no = mach_read_from_4(
page + FIL_PAGE_NEXT);
mtr.write<4>(*block, BTR_EXTERN_PAGE_NO + field_ref,
next_page_no); /* Zero out the BLOB length. If the server crashesduringtheexecutionofthisfunction, trx_rollback_all_recovered()could dereferencethehalf-deletedBLOB,fetchinga
wrong prefix for the BLOB. */
mtr.write<4,mtr_t::MAYBE_NOP>(*block,
BTR_EXTERN_LEN + 4
+ field_ref, 0U);
}
/* Commit mtr and release the BLOB block to save memory. */
btr_blob_free(ext_block, TRUE, &mtr);
}
}
/***********************************************************//**
Frees the externally stored fields for a record. */ static void
btr_rec_free_externally_stored_fields( /*==================================*/
dict_index_t* index, /*!< in: index of the data, the index
tree MUST be X-latched */
rec_t* rec, /*!< in/out: record */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec, index) */
buf_block_t* block, /*!< in: index page of rec */ bool rollback,/*!< in: performing rollback? */
mtr_t* mtr) /*!< in: mini-transaction handle which contains anX-latchtorecordpageandtotheindex
tree */
{
ulint n_fields;
ulint i;
ut_ad(rec_offs_validate(rec, index, offsets));
ut_ad(mtr->memo_contains_page_flagged(rec, MTR_MEMO_PAGE_X_FIX));
ut_ad(index->is_primary());
ut_ad(page_rec_is_leaf(rec)); /* Free possible externally stored fields in the record */
for (i = 0; i < n_fields; i++) { if (rec_offs_nth_extern(offsets, i)) {
btr_free_externally_stored_field(
index, btr_rec_get_field_ref(rec, offsets, i),
rec, offsets, block, i, rollback, mtr);
}
}
}
/***********************************************************//**
Frees the externally stored fields for a record, if the field is mentioned
in the update vector. */ static void
btr_rec_free_updated_extern_fields( /*===============================*/
dict_index_t* index, /*!< in: index of rec; the index tree MUST be
X-latched */
rec_t* rec, /*!< in/out: record */
buf_block_t* block, /*!< in: index page of rec */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec, index) */ const upd_t* update, /*!< in: update vector */ bool rollback,/*!< in: performing rollback? */
mtr_t* mtr) /*!< in: mini-transaction handle which contains
an X-latch to record page and to the tree */
{
ulint n_fields;
ulint i;
/* Free possible externally stored fields in the record */
n_fields = upd_get_n_fields(update);
for (i = 0; i < n_fields; i++) { const upd_field_t* ufield = upd_get_nth_field(update, i);
if (rec_offs_nth_extern(offsets, ufield->field_no)) {
ulint len;
byte* data = rec_get_nth_field(
rec, offsets, ufield->field_no, &len);
ut_a(len >= BTR_EXTERN_FIELD_REF_SIZE);
btr_free_externally_stored_field(
index, data + len - BTR_EXTERN_FIELD_REF_SIZE,
rec, offsets, block,
ufield->field_no, rollback, mtr);
}
}
}
/*******************************************************************//**
Copies the prefix of an uncompressed BLOB. The clustered index record
that points to this BLOB must be protected by a lock or a page latch.
@return number of bytes written to buf */ static
ulint
btr_copy_blob_prefix( /*=================*/
byte* buf, /*!< out: the externally stored part of
the field, or a prefix of it */
uint32_t len, /*!< in: length of buf, in bytes */
page_id_t id, /*!< in: page identifier of the first BLOB page */
uint32_t offset) /*!< in: offset on the first BLOB page */
{
ulint copied_len = 0;
THD* thd{current_thd};
/* Zlib inflate needs 32 kilobytes for the default
window size, plus a few kilobytes for small objects. */
heap = mem_heap_create(40000);
page_zip_set_alloc(&d_stream, heap);
/** Copies the prefix of an externally stored field of a record. TheclusteredindexrecordthatpointstothisBLOBmustbeprotected byalockorapagelatch. @param[out]buftheexternallystoredpartofthe field,oraprefixofit @param[in]lenlengthofbuf,inbytes @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]idpageidentifierofthefirstBLOBpage @param[in]offsetoffsetonthefirstBLOBpage
@return number of bytes written to buf */ static
ulint
btr_copy_externally_stored_field_prefix_low(
byte* buf,
uint32_t len,
ulint zip_size,
page_id_t id,
uint32_t offset)
{ if (len == 0) return0;
/** Copies the prefix of an externally stored field of a record. Theclusteredindexrecordmustbeprotectedbyalockorapagelatch. @param[out]bufthefield,oraprefixofit @param[in]lenlengthofbuf,inbytes @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]data'internally'storedpartofthefield containingalsothereferencetotheexternalpart;mustbeprotectedby alockorapagelatch @param[in]local_lenlengthofdata,inbytes @returnthelengthofthecopiedfield,or0ifthecolumnwasbeing
or has been deleted */
ulint
btr_copy_externally_stored_field_prefix(
byte* buf,
ulint len,
ulint zip_size, const byte* data,
ulint local_len)
{
ut_a(local_len >= BTR_EXTERN_FIELD_REF_SIZE);
local_len -= BTR_EXTERN_FIELD_REF_SIZE;
if (UNIV_UNLIKELY(local_len >= len)) {
memcpy(buf, data, len); return(len);
}
if (!mach_read_from_4(data + BTR_EXTERN_LEN + 4)) { /* The externally stored part of the column has been (partially)deleted.Signalthehalf-deletedBLOB
to the caller. */
/** Copies an externally stored field of a record to mem heap. Theclusteredindexrecordmustbeprotectedbyalockorapagelatch. @param[out]lenlengthofthewholefield @param[in]data'internally'storedpartofthefield containingalsothereferencetotheexternalpart;mustbeprotectedby alockorapagelatch @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]local_lenlengthofdata @param[in,out]heapmemheap
@return the whole field copied to heap */
byte*
btr_copy_externally_stored_field(
ulint* len, const byte* data,
ulint zip_size,
ulint local_len,
mem_heap_t* heap)
{
byte* buf;
/** Copies an externally stored field of a record to mem heap. @param[in]recrecordinaclusteredindex;mustbe protectedbyalockorapagelatch @param[in]offsetarrayreturnedbyrec_get_offsets() @param[in]zip_sizeROW_FORMAT=COMPRESSEDpagesize,or0 @param[in]nofieldnumber @param[out]lenlengthofthefield @param[in,out]heapmemheap
@return the field copied to heap, or NULL if the field is incomplete */
byte*
btr_rec_copy_externally_stored_field( const rec_t* rec, const rec_offs* offsets,
ulint zip_size,
ulint no,
ulint* len,
mem_heap_t* heap)
{
ulint local_len; const byte* data;
ut_a(rec_offs_nth_extern(offsets, no));
/* An externally stored field can contain some initial datafromthefield,andinthelast20bytesithasthe spaceid,pagenumber,andoffsetwheretherestofthe fielddataisstored,andthedatalengthinadditionto thedatastoredlocally.Wemayneedtostoresomedata locallytogetthelocalrecordlengthabovethe128byte limitsothatfieldoffsetsarestoredintwobytes,and
the extern bit is available in those two bytes. */
data = rec_get_nth_field(rec, offsets, no, &local_len);
ut_a(local_len >= BTR_EXTERN_FIELD_REF_SIZE);
if (UNIV_UNLIKELY
(!memcmp(data + local_len - BTR_EXTERN_FIELD_REF_SIZE,
field_ref_zero, BTR_EXTERN_FIELD_REF_SIZE))) { /* The externally stored field was not written yet. Thisrecordshouldonlybeseenby trx_rollback_recovered()orany
TRX_ISO_READ_UNCOMMITTED transactions. */ return(NULL);
}
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.472Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.