/** Statistics on compression, indexed by page_zip_des_t::ssize - 1 */
page_zip_stat_t page_zip_stat[PAGE_ZIP_SSIZE_MAX]; /** Statistics on compression, indexed by index->id */
page_zip_stat_per_index_t page_zip_stat_per_index;
/** Compression level to be used by zlib. Settable by user. */
uint page_zip_level;
/* Please refer to ../include/page0zip.ic for a description of the
compressed page format. */
/* The infimum and supremum records are omitted from the compressed page. Oncompress,wecomparethattherecordsarethere,andonuncompresswe
restore the records. */ /** Extra bytes of an infimum record */ staticconst byte infimum_extra[] = { 0x01, /* info_bits=0, n_owned=1 */ 0x00, 0x02 /* heap_no=0, status=2 */ /* ?, ? */ /* next=(first user rec, or supremum) */
}; /** Data bytes of an infimum record */ staticconst byte infimum_data[] = { 0x69, 0x6e, 0x66, 0x69, 0x6d, 0x75, 0x6d, 0x00 /* "infimum\0" */
}; /** Extra bytes and data bytes of a supremum record */ staticconst byte supremum_extra_data alignas(4) [] = { /* 0x0?, */ /* info_bits=0, n_owned=1..8 */ 0x00, 0x0b, /* heap_no=1, status=3 */ 0x00, 0x00, /* next=0 */ 0x73, 0x75, 0x70, 0x72, 0x65, 0x6d, 0x75, 0x6d /* "supremum" */
};
/** Assert that a block of memory is filled with zero bytes. @parambin:memoryblock
@param s in: size of the memory block, in bytes */ #define ASSERT_ZERO(b, s) ut_ad(!memcmp(b, field_ref_zero, s)) /** Assert that a BLOB pointer is filled with zero bytes.
@param b in: BLOB pointer */ #define ASSERT_ZERO_BLOB(b) ASSERT_ZERO(b, FIELD_REF_SIZE)
/* Enable some extra debugging output. This code can be enabled
independently of any UNIV_ debugging conditions. */ #ifdefined UNIV_DEBUG || defined UNIV_ZIP_DEBUG # include <stdarg.h>
MY_ATTRIBUTE((format (printf, 1, 2))) /**********************************************************************//**
Report a failure to decompress or compress.
@return number of characters printed */ static int
page_zip_fail_func( /*===============*/ constchar* fmt, /*!< in: printf(3) format string */
...) /*!< in: arguments corresponding to fmt */
{ int res;
va_list ap;
return(res);
} /** Wrapper for page_zip_fail_func()
@param fmt_args in: printf(3) format string and arguments */ # define page_zip_fail(fmt_args) page_zip_fail_func fmt_args #else/* UNIV_DEBUG || UNIV_ZIP_DEBUG */ /** Dummy wrapper for page_zip_fail_func()
@param fmt_args ignored: printf(3) format string and arguments */ # define page_zip_fail(fmt_args) /* empty */ #endif/* UNIV_DEBUG || UNIV_ZIP_DEBUG */
/**********************************************************************//**
Determine the guaranteed free space on an empty page.
@return minimum payload size on the page */
ulint
page_zip_empty_size( /*================*/
ulint n_fields, /*!< in: number of columns in the index */
ulint zip_size) /*!< in: compressed page size in bytes */
{
ulint size = zip_size /* subtract the page header and the longest
uncompressed data needed for one record */
- (PAGE_DATA
+ PAGE_ZIP_CLUST_LEAF_SLOT_SIZE
+ 1/* encoded heap_no==2 in page_zip_write_rec() */
+ 1/* end of modification log */
- REC_N_NEW_EXTRA_BYTES/* omitted bytes */) /* subtract the space for page_zip_fields_encode() */
- compressBound(static_cast<uLong>(2 * (n_fields + 1))); return(lint(size) > 0 ? size : 0);
}
/** Check whether a tuple is too big for compressed table @param[in]indexdictindexobject @param[in]entryentryfortheindex
@return true if it's too big, otherwise false */ bool
page_zip_is_too_big( const dict_index_t* index, const dtuple_t* entry)
{ const ulint zip_size = index->table->space->zip_size();
/* Estimate the free space of an empty compressed page. Subtractonebytefortheencodedheap_nointhe
modification log. */
ulint free_space_zip = page_zip_empty_size(
index->n_fields, zip_size);
ulint n_uniq = dict_index_get_n_unique_in_tree(index);
/* Subtract one byte for the encoded heap_no in the
modification log. */
free_space_zip--;
/* There should be enough room for two node pointer recordsonanemptynon-leafpage.Thisprevents
infinite page splits. */
if (entry->n_fields >= n_uniq
&& (REC_NODE_PTR_SIZE
+ rec_get_converted_size_comp_prefix(
index, entry->fields, n_uniq, NULL) /* On a compressed page, there is atwo-byteentryinthedense pagedirectoryforeveryrecord.
But there is no record header. */
- (REC_N_NEW_EXTRA_BYTES - 2)
> free_space_zip / 2)) { return(true);
}
return(false);
}
/*************************************************************//**
Gets the number of elements in the dense page directory,
including deleted records (the free list).
@return number of elements in the dense page directory */
UNIV_INLINE
ulint
page_zip_dir_elems( /*===============*/ const page_zip_des_t* page_zip) /*!< in: compressed page */
{ /* Exclude the page infimum and supremum from the record count. */ return ulint(page_dir_get_n_heap(page_zip->data))
- PAGE_HEAP_NO_USER_LOW;
}
/*************************************************************//**
Gets the size of the compressed page trailer (the dense page directory),
including deleted records (the free list).
@return length of dense page directory, in bytes */
UNIV_INLINE
ulint
page_zip_dir_size( /*==============*/ const page_zip_des_t* page_zip) /*!< in: compressed page */
{ return(PAGE_ZIP_DIR_SLOT_SIZE * page_zip_dir_elems(page_zip));
}
/*************************************************************//**
Gets an offset to the compressed page trailer (the dense page directory),
including deleted records (the free list).
@return offset of the dense page directory */
UNIV_INLINE
ulint
page_zip_dir_start_offs( /*====================*/ const page_zip_des_t* page_zip, /*!< in: compressed page */
ulint n_dense) /*!< in: directory size */
{
ut_ad(n_dense * PAGE_ZIP_DIR_SLOT_SIZE < page_zip_get_size(page_zip));
/*************************************************************//**
Gets a pointer to the compressed page trailer (the dense page directory),
including deleted records (the free list).
@param[in] page_zip compressed page
@param[in] n_dense number of entries in the directory
@return pointer to the dense page directory */ #define page_zip_dir_start_low(page_zip, n_dense) \
((page_zip)->data + page_zip_dir_start_offs(page_zip, n_dense)) /*************************************************************//**
Gets a pointer to the compressed page trailer (the dense page directory),
including deleted records (the free list).
@param[in] page_zip compressed page
@return pointer to the dense page directory */ #define page_zip_dir_start(page_zip) \
page_zip_dir_start_low(page_zip, page_zip_dir_elems(page_zip))
/*************************************************************//**
Gets the size of the compressed page trailer (the dense page directory),
only including user records (excluding the free list).
@return length of dense page directory comprising existing records, in bytes */
UNIV_INLINE
ulint
page_zip_dir_user_size( /*===================*/ const page_zip_des_t* page_zip) /*!< in: compressed page */
{
ulint size = PAGE_ZIP_DIR_SLOT_SIZE
* ulint(page_get_n_recs(page_zip->data));
ut_ad(size <= page_zip_dir_size(page_zip)); return(size);
}
/*************************************************************//**
Find the slot of the given record in the dense page directory.
@return dense directory slot, or NULL if record not found */
UNIV_INLINE
byte*
page_zip_dir_find_low( /*==================*/
byte* slot, /*!< in: start of records */
byte* end, /*!< in: end of records */
ulint offset) /*!< in: offset of user record */
{
ut_ad(slot <= end);
for (; slot < end; slot += PAGE_ZIP_DIR_SLOT_SIZE) { if ((mach_read_from_2(slot) & PAGE_ZIP_DIR_SLOT_MASK)
== offset) { return(slot);
}
}
return(NULL);
}
/*************************************************************//**
Find the slot of the given non-free record in the dense page directory.
@return dense directory slot, or NULL if record not found */
UNIV_INLINE
byte*
page_zip_dir_find( /*==============*/
page_zip_des_t* page_zip, /*!< in: compressed page */
ulint offset) /*!< in: offset of user record */
{
byte* end = page_zip->data + page_zip_get_size(page_zip);
/*************************************************************//**
Find the slot of the given free record in the dense page directory.
@return dense directory slot, or NULL if record not found */
UNIV_INLINE
byte*
page_zip_dir_find_free( /*===================*/
page_zip_des_t* page_zip, /*!< in: compressed page */
ulint offset) /*!< in: offset of user record */
{
byte* end = page_zip->data + page_zip_get_size(page_zip);
ut_ad(page_zip_simple_validate(page_zip));
return(page_zip_dir_find_low(end - page_zip_dir_size(page_zip),
end - page_zip_dir_user_size(page_zip),
offset));
}
/*************************************************************//**
Read a given slot in the dense page directory.
@return record offset on the uncompressed page, possibly ORed with
PAGE_ZIP_DIR_SLOT_DEL or PAGE_ZIP_DIR_SLOT_OWNED */
UNIV_INLINE
ulint
page_zip_dir_get( /*=============*/ const page_zip_des_t* page_zip, /*!< in: compressed page */
ulint slot) /*!< in: slot
(0=first user record) */
{
ut_ad(page_zip_simple_validate(page_zip));
ut_ad(slot < page_zip_dir_size(page_zip) / PAGE_ZIP_DIR_SLOT_SIZE); return(mach_read_from_2(page_zip->data + page_zip_get_size(page_zip)
- PAGE_ZIP_DIR_SLOT_SIZE * (slot + 1)));
}
/** Write a byte string to a ROW_FORMAT=COMPRESSED page. @param[in]bROW_FORMAT=COMPRESSEDindexpage @param[in]offsetbyteoffsetfromb.zip.data
@param[in] len length of the data to write */ inlinevoid mtr_t::zmemcpy(const buf_block_t &b, ulint offset, ulint len)
{
ut_ad(fil_page_get_type(b.page.zip.data) == FIL_PAGE_INDEX ||
fil_page_get_type(b.page.zip.data) == FIL_PAGE_RTREE);
ut_ad(page_zip_simple_validate(&b.page.zip));
ut_ad(offset + len <= page_zip_get_size(&b.page.zip));
heap_no = rec_get_heap_no_new(rec);
ut_ad(heap_no >= PAGE_HEAP_NO_USER_LOW);
left = heap_no - PAGE_HEAP_NO_USER_LOW; if (UNIV_UNLIKELY(!left)) { return(0);
}
for (i = 0; i < n_recs; i++) { const rec_t* r = page + (page_zip_dir_get(page_zip, i)
& PAGE_ZIP_DIR_SLOT_MASK);
if (rec_get_heap_no_new(r) < heap_no) {
n_ext += rec_get_n_extern_new(r, index,
ULINT_UNDEFINED); if (!--left) { break;
}
}
}
return(n_ext);
}
/**********************************************************************//**
Encode the length of a fixed-length column.
@return buf + length of encoded val */ static
byte*
page_zip_fixed_field_encode( /*========================*/
byte* buf, /*!< in: pointer to buffer where to write */
ulint val) /*!< in: value to write */
{
ut_ad(val >= 2);
/**********************************************************************//**
Write the index information for the compressed page.
@return used size of buf */
ulint
page_zip_fields_encode( /*===================*/
ulint n, /*!< in: number of fields
to compress */ const dict_index_t* index, /*!< in: index comprising
at least n fields */
ulint trx_id_pos, /*!< in: position of the trx_id column intheindex,orULINT_UNDEFINEDif
this is a non-leaf page */
byte* buf) /*!< out: buffer of (n + 1) * 2 bytes */
{ const byte* buf_start = buf;
ulint i;
ulint col;
ulint trx_id_col = 0; /* sum of lengths of preceding non-nullable fixed fields, or 0 */
ulint fixed_sum = 0;
if (fixed_sum && UNIV_UNLIKELY
(fixed_sum + field->fixed_len
> DICT_MAX_FIXED_COL_LEN)) { /* Write out the length of the precedingnon-nullablefields, toavoidexceedingthemaximum
length of a fixed-length column. */
buf = page_zip_fixed_field_encode(
buf, fixed_sum << 1 | 1);
fixed_sum = 0;
col++;
}
if (i && UNIV_UNLIKELY(i == trx_id_pos)) { if (fixed_sum) { /* Write out the length of any precedingnon-nullablefields,
and start a new trx_id column. */
buf = page_zip_fixed_field_encode(
buf, fixed_sum << 1 | 1);
col++;
}
trx_id_col = col;
fixed_sum = field->fixed_len;
} else { /* add to the sum */
fixed_sum += field->fixed_len;
}
} else { /* fixed-length nullable field */
if (fixed_sum) { /* write out the length of any
preceding non-nullable fields */
buf = page_zip_fixed_field_encode(
buf, fixed_sum << 1 | 1);
fixed_sum = 0;
col++;
}
if (fixed_sum) { /* Write out the lengths of last fixed-length columns. */
buf = page_zip_fixed_field_encode(buf, fixed_sum << 1 | 1);
}
if (trx_id_pos != ULINT_UNDEFINED) { /* Write out the position of the trx_id column */
i = trx_id_col;
} else { /* Write out the number of nullable fields */
i = index->n_nullable;
}
if (i < 128) {
*buf++ = (byte) i;
} else {
*buf++ = (byte) (0x80 | i >> 8);
*buf++ = (byte) i;
}
/**********************************************************************//**
Populate the dense page directory from the sparse directory. */ static void
page_zip_dir_encode( /*================*/ const page_t* page, /*!< in: compact page */
byte* buf, /*!< in: pointer to dense page directory[-1];
out: dense directory on compressed page */ const rec_t** recs) /*!< in: pointer to an array of 0, or NULL; out:densepagedirectorysortedbyascending
address (and heap_no) */
{ const byte* rec;
ulint status;
ulint min_mark;
ulint heap_no;
ulint i;
ulint n_heap;
ulint offs;
min_mark = 0;
if (page_is_leaf(page)) {
status = REC_STATUS_ORDINARY;
} else {
status = REC_STATUS_NODE_PTR; if (UNIV_UNLIKELY(!page_has_prev(page))) {
min_mark = REC_INFO_MIN_REC_FLAG;
}
}
n_heap = page_dir_get_n_heap(page);
/* Traverse the list of stored records in the collation order,
starting from the first user record. */
if (UNIV_UNLIKELY(rec_get_n_owned_new(rec) != 0)) {
offs |= PAGE_ZIP_DIR_SLOT_OWNED;
}
info_bits = rec_get_info_bits(rec, TRUE); if (info_bits & REC_INFO_DELETED_FLAG) {
info_bits &= ~REC_INFO_DELETED_FLAG;
offs |= PAGE_ZIP_DIR_SLOT_DEL;
}
ut_a(info_bits == min_mark); /* Only the smallest user record can have
REC_INFO_MIN_REC_FLAG set. */
min_mark = 0;
if (UNIV_LIKELY_NULL(recs)) { /* Ensure that each heap_no occurs at most once. */
ut_a(!recs[heap_no - PAGE_HEAP_NO_USER_LOW]); /* exclude infimum and supremum */
recs[heap_no - PAGE_HEAP_NO_USER_LOW] = rec;
}
ut_a(ulint(rec_get_status(rec)) == status);
}
offs = page_header_get_field(page, PAGE_FREE);
/* Traverse the free list (of deleted records). */ while (offs) {
ut_ad(!(offs & ~PAGE_ZIP_DIR_SLOT_MASK));
rec = page + offs;
if (UNIV_LIKELY_NULL(recs)) { /* Ensure that each heap_no occurs at most once. */
ut_a(!recs[heap_no - PAGE_HEAP_NO_USER_LOW]); /* exclude infimum and supremum */
recs[heap_no - PAGE_HEAP_NO_USER_LOW] = rec;
}
offs = rec_get_next_offs(rec, TRUE);
}
/* Ensure that each heap no occurs at least once. */
ut_a(i + PAGE_HEAP_NO_USER_LOW == n_heap);
}
extern"C" {
/**********************************************************************//**
Allocate memory for zlib. */ static void*
page_zip_zalloc( /*============*/ void* opaque, /*!< in/out: memory heap */
uInt items, /*!< in: number of items to allocate */
uInt size) /*!< in: size of an item in bytes */
{ return(mem_heap_zalloc(static_cast<mem_heap_t*>(opaque), items * size));
}
#if0 || defined UNIV_DEBUG || defined UNIV_ZIP_DEBUG /** Symbol for enabling compression and decompression diagnostics */ # define PAGE_ZIP_COMPRESS_DBG #endif
#ifdef PAGE_ZIP_COMPRESS_DBG /** Set this variable in a debugger to enable
excessive logging in page_zip_compress(). */ staticbool page_zip_compress_dbg; /** Set this variable in a debugger to enable binaryloggingofthedatapassedtodeflate(). Whenthisvariableisnonzero,itwillact
as a log file name generator. */ staticunsigned page_zip_compress_log;
/**********************************************************************//**
Wrapper for deflate(). Log the operation if page_zip_compress_dbg is set.
@return deflate() status: Z_OK, Z_BUF_ERROR, ... */ static int
page_zip_compress_deflate( /*======================*/
FILE* logfile,/*!< in: log file, or NULL */
z_streamp strm, /*!< in/out: compressed stream for deflate() */ int flush) /*!< in: deflate() flushing method */
{ int status; if (UNIV_UNLIKELY(page_zip_compress_dbg)) {
ut_print_buf(stderr, strm->next_in, strm->avail_in);
} if (UNIV_LIKELY_NULL(logfile)) { if (fwrite(strm->next_in, 1, strm->avail_in, logfile)
!= strm->avail_in) {
perror("fwrite");
}
}
status = deflate(strm, flush); if (UNIV_UNLIKELY(page_zip_compress_dbg)) {
fprintf(stderr, " -> %d\n", status);
} return(status);
}
/**********************************************************************//**
Compress the records of a leaf node of a secondary index.
@return Z_OK, or a zlib error code */ static int
page_zip_compress_sec( /*==================*/
FILE_LOGFILE
z_stream* c_stream, /*!< in/out: compressed page stream */ const rec_t** recs, /*!< in: dense page directory
sorted by address */
ulint n_dense) /*!< in: size of recs[] */
{ int err = Z_OK;
ut_ad(n_dense > 0);
do { const rec_t* rec = *recs++;
/* Compress everything up to this record. */
c_stream->avail_in = static_cast<uInt>(
rec - REC_N_NEW_EXTRA_BYTES
- c_stream->next_in);
if (UNIV_LIKELY(c_stream->avail_in != 0)) {
MEM_CHECK_DEFINED(c_stream->next_in,
c_stream->avail_in);
err = deflate(c_stream, Z_NO_FLUSH); if (UNIV_UNLIKELY(err != Z_OK)) { break;
}
}
c_stream->next_in = (byte*) rec;
} while (--n_dense);
return(err);
}
/**********************************************************************//**
Compress a record of a leaf node of a clustered index that contains
externally stored columns.
@return Z_OK, or a zlib error code */ static int
page_zip_compress_clust_ext( /*========================*/
FILE_LOGFILE
z_stream* c_stream, /*!< in/out: compressed page stream */ const rec_t* rec, /*!< in: record */ const rec_offs* offsets, /*!< in: rec_get_offsets(rec) */
ulint trx_id_col, /*!< in: position of of DB_TRX_ID */
byte* deleted, /*!< in: dense directory entry pointing
to the head of the free list */
byte* storage, /*!< in: end of dense page directory */
byte** externs, /*!< in/out: pointer to the next
available BLOB pointer */
ulint* n_blobs) /*!< in/out: number of
externally stored columns */
{ int err;
ulint i;
/**********************************************************************//**
Compress the records of a leaf node of a clustered index.
@return Z_OK, or a zlib error code */ static int
page_zip_compress_clust( /*====================*/
FILE_LOGFILE
z_stream* c_stream, /*!< in/out: compressed page stream */ const rec_t** recs, /*!< in: dense page directory
sorted by address */
ulint n_dense, /*!< in: size of recs[] */
dict_index_t* index, /*!< in: the index of the page */
ulint* n_blobs, /*!< in: 0; out: number of
externally stored columns */
ulint trx_id_col, /*!< index of the trx_id column */
byte* deleted, /*!< in: dense directory entry pointing
to the head of the free list */
byte* storage, /*!< in: end of dense page directory */
mem_heap_t* heap) /*!< in: temporary memory heap */
{ int err = Z_OK;
rec_offs* offsets = NULL; /* BTR_EXTERN_FIELD_REF storage */
byte* externs = storage - n_dense
* (DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN);
/* Check if there are any externally stored columns. Foreachexternallystoredcolumn,storethe
BTR_EXTERN_FIELD_REF separately. */ if (rec_offs_any_extern(offsets)) {
ut_ad(dict_index_is_clust(index));
/* The dense directory excludes the infimum and supremum records. */
n_dense = ulint(page_dir_get_n_heap(page)) - PAGE_HEAP_NO_USER_LOW; #ifdef PAGE_ZIP_COMPRESS_DBG if (UNIV_UNLIKELY(page_zip_compress_dbg)) {
ib::info() << "compress "
<< static_cast<void*>(page_zip) << " "
<< static_cast<constvoid*>(page) << " "
<< page_is_leaf(page) << " "
<< n_fields << " " << n_dense;
}
if (UNIV_UNLIKELY(page_zip_compress_log)) { /* Create a log file for every compression attempt. */ char logfilename[9];
snprintf(logfilename, sizeof logfilename, "%08x", page_zip_compress_log++);
logfile = fopen(logfilename, "wb");
if (logfile) { /* Write the uncompressed page to the log. */ if (fwrite(page, 1, srv_page_size, logfile)
!= srv_page_size) {
perror("fwrite");
} /* Record the compressed size as zero.
This will be overwritten at successful exit. */
putc(0, logfile);
putc(0, logfile);
putc(0, logfile);
putc(0, logfile);
}
} #endif/* PAGE_ZIP_COMPRESS_DBG */
page_zip_stat[page_zip->ssize - 1].compressed++; if (cmp_per_index_enabled) {
mysql_mutex_lock(&page_zip_stat_per_index_mutex);
page_zip_stat_per_index[ind_id].compressed++;
mysql_mutex_unlock(&page_zip_stat_per_index_mutex);
}
if (UNIV_UNLIKELY(n_dense * PAGE_ZIP_DIR_SLOT_SIZE
>= page_zip_get_size(page_zip))) {
/* Subtract the space reserved for uncompressed data. */ /* Page header and the end marker of the modification log */
c_stream.avail_out = static_cast<uInt>(buf_end - buf - 1);
/* Dense page directory and uncompressed columns, if any */ if (page_is_leaf(page)) { if (dict_index_is_clust(index)) {
trx_id_col = index->db_trx_id();
/* Compress the records in heap_no order. */ if (UNIV_UNLIKELY(!n_dense)) {
} elseif (!page_is_leaf(page)) { /* This is a node pointer page. */
err = page_zip_compress_node_ptrs(LOGFILE
&c_stream, recs, n_dense,
index, storage, heap); if (UNIV_UNLIKELY(err != Z_OK)) { goto zlib_error;
}
} elseif (UNIV_LIKELY(trx_id_col == ULINT_UNDEFINED)) { /* This is a leaf page in a secondary index. */
err = page_zip_compress_sec(LOGFILE
&c_stream, recs, n_dense); if (UNIV_UNLIKELY(err != Z_OK)) { goto zlib_error;
}
} else { /* This is a leaf page in a clustered index. */
err = page_zip_compress_clust(LOGFILE
&c_stream, recs, n_dense,
index, &n_blobs, trx_id_col,
buf_end - PAGE_ZIP_DIR_SLOT_SIZE
* page_get_n_recs(page),
storage, heap); if (UNIV_UNLIKELY(err != Z_OK)) { goto zlib_error;
}
}
/* Finish the compression. */
ut_ad(!c_stream.avail_in); /* Compress any trailing garbage, in case the last record was allocatedfromanoriginallylongerspaceonthefreelist,
or the data of the last record from page_zip_compress_sec(). */
c_stream.avail_in = static_cast<uInt>(
page_header_get_field(page, PAGE_HEAP_TOP)
- (c_stream.next_in - page));
ut_a(c_stream.avail_in <= srv_page_size - PAGE_ZIP_START - PAGE_DIR);
#ifdefined HAVE_valgrind && !__has_feature(memory_sanitizer) /* Valgrind believes that zlib does not initialize some bits
in the last 7 or 8 bytes of the stream. Make Valgrind happy. */
MEM_MAKE_DEFINED(buf, c_stream.total_out); #endif/* HAVE_valgrind && !memory_sanitizer */
/* Zero out the area reserved for the modification log. Spacefortheendmarkerofthemodificationlogisnot
included in avail_out. */
memset(c_stream.next_out, 0, c_stream.avail_out + 1/* end marker */);
#ifdef UNIV_DEBUG
page_zip->m_start = #endif/* UNIV_DEBUG */
page_zip->m_end = uint16_t(PAGE_DATA + c_stream.total_out);
page_zip->m_nonempty = FALSE;
page_zip->n_blobs = unsigned(n_blobs) & ((1U << 12) - 1); /* Copy those header fields that will not be written
in buf_flush_init_for_writing() */
memcpy_aligned<8>(page_zip->data + FIL_PAGE_PREV, page + FIL_PAGE_PREV,
FIL_PAGE_LSN - FIL_PAGE_PREV);
memcpy_aligned<2>(page_zip->data + FIL_PAGE_TYPE, page + FIL_PAGE_TYPE, 2);
memcpy_aligned<2>(page_zip->data + FIL_PAGE_DATA, page + FIL_PAGE_DATA,
PAGE_DATA - FIL_PAGE_DATA); /* Copy the rest of the compressed page */
memcpy_aligned<2>(page_zip->data + PAGE_DATA, buf,
page_zip_get_size(page_zip) - PAGE_DATA);
mem_heap_free(heap); #ifdef UNIV_ZIP_DEBUG
ut_a(page_zip_validate(page_zip, page, index)); #endif/* UNIV_ZIP_DEBUG */
#ifdef PAGE_ZIP_COMPRESS_DBG if (logfile) { /* Record the compressed size of the block. */
byte sz[4];
mach_write_to_4(sz, c_stream.total_out);
fseek(logfile, srv_page_size, SEEK_SET); if (fwrite(sz, 1, sizeof sz, logfile) != sizeof sz) {
perror("fwrite");
}
fclose(logfile);
} #endif/* PAGE_ZIP_COMPRESS_DBG */ const uint64_t time_diff = (my_interval_timer() - ns) / 1000;
page_zip_stat[page_zip->ssize - 1].compressed_ok++;
page_zip_stat[page_zip->ssize - 1].compressed_usec += time_diff; if (cmp_per_index_enabled) {
mysql_mutex_lock(&page_zip_stat_per_index_mutex);
page_zip_stat_per_index[ind_id].compressed_ok++;
page_zip_stat_per_index[ind_id].compressed_usec += time_diff;
mysql_mutex_unlock(&page_zip_stat_per_index_mutex);
}
if (page_is_leaf(page)) {
dict_index_zip_success(index);
}
returntrue;
}
/**********************************************************************//**
Deallocate the index information initialized by page_zip_fields_decode(). */ static void
page_zip_fields_free( /*=================*/
dict_index_t* index) /*!< in: dummy index to be freed */
{ if (index) {
dict_table_t* table = index->table;
index->zip_pad.mutex.~mutex();
mem_heap_free(index->heap);
dict_mem_table_free(table);
}
}
/**********************************************************************//**
Read the index information for the compressed page.
@return own: dummy index describing the page, or NULL on error */ static
dict_index_t*
page_zip_fields_decode( /*===================*/ const byte* buf, /*!< in: index information */ const byte* end, /*!< in: end of buf */
ulint* trx_id_col,/*!< in: NULL for non-leaf pages; forleafpages,pointertowheretostore
the position of the trx_id column */ bool is_spatial)/*< in: is spatial index or not */
{ const byte* b;
ulint n;
ulint i;
ulint val;
dict_table_t* table;
dict_index_t* index;
/* Determine the number of fields. */ for (b = buf, n = 0; b < end; n++) { if (*b++ & 0x80) {
b++; /* skip the second byte */
}
}
n--; /* n_nullable or trx_id */
if (UNIV_UNLIKELY(n > REC_MAX_N_FIELDS)) {
page_zip_fail(("page_zip_fields_decode: n = %lu\n",
(ulong) n)); return(NULL);
}
val = *b++; if (UNIV_UNLIKELY(val & 0x80)) {
val = (val & 0x7f) << 8 | *b++;
}
/* Decode the position of the trx_id column. */ if (trx_id_col) { if (!val) {
val = ULINT_UNDEFINED;
} elseif (UNIV_UNLIKELY(val >= n)) {
fail:
page_zip_fields_free(index); return NULL;
} else {
index->type = DICT_CLUSTERED;
}
*trx_id_col = val;
} else { /* Decode the number of nullable fields. */ if (UNIV_UNLIKELY(index->n_nullable > val)) { goto fail;
} else {
index->n_nullable = static_cast<unsigned>(val)
& dict_index_t::MAX_N_FIELDS;
}
}
/* ROW_FORMAT=COMPRESSED does not support instant ADD COLUMN */
index->n_core_fields = index->n_fields;
index->n_core_null_bytes = static_cast<uint8_t>(
UT_BITS_IN_BYTES(unsigned(index->n_nullable)));
ut_ad(b == end);
if (is_spatial) {
index->type |= DICT_SPATIAL;
}
return(index);
}
/**********************************************************************//**
Populate the sparse page directory from the dense directory.
@returnTRUE on success, FALSE on failure */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
ibool
page_zip_dir_decode( /*================*/ const page_zip_des_t* page_zip,/*!< in: dense page directory on
compressed page */
page_t* page, /*!< in: compact page with valid header; out:trailerandsparsepagedirectory
filled in */
rec_t** recs, /*!< out: dense page directory sorted by
ascending address (and heap_no) */
ulint n_dense)/*!< in: number of user records, and
size of recs[] */
{
ulint i;
ulint n_recs;
byte* slot;
memcpy(next_out, data, len);
data += len;
next_out += len
+ BTR_EXTERN_FIELD_REF_SIZE;
}
}
/* Copy the last bytes of the record. */
len = ulint(rec_get_end(rec, offsets) - next_out); if (UNIV_UNLIKELY(data + len >= end)) {
page_zip_fail(("page_zip_apply_log_ext:" " last %p+%lu >= %p\n",
(constvoid*) data,
(ulong) len,
(constvoid*) end)); return(NULL);
}
memcpy(next_out, data, len);
data += len;
return(data);
}
/**********************************************************************//**
Apply the modification log to an uncompressed page. Donot copy the fields that are stored separately.
@return pointer to end of modification log, or NULL on failure */ static const byte*
page_zip_apply_log( /*===============*/ const byte* data, /*!< in: modification log */
ulint size, /*!< in: maximum length of the log, in bytes */
rec_t** recs, /*!< in: dense page directory, sortedbyaddress(indexedby
heap_no - PAGE_HEAP_NO_USER_LOW) */
ulint n_dense,/*!< in: size of recs[] */
ulint n_core, /*!< in: index->n_fields, or 0 for non-leaf */
ulint trx_id_col,/*!< in: column number of trx_id in the index,
or ULINT_UNDEFINED if none */
ulint heap_status, /*!< in: heap_no and status bits for
the next record to uncompress */
dict_index_t* index, /*!< in: index of the page */
rec_offs* offsets)/*!< in/out: work area for
rec_get_offsets_reverse() */
{ const byte* const end = data + size;
/* This may either be an old record that is being overwritten(updatedinplace,orallocatedfrom thefreelist),oranewrecord,withthenext
available_heap_no. */ if (UNIV_UNLIKELY(hs > heap_status)) {
page_zip_fail(("page_zip_apply_log: %lu > %lu\n",
(ulong) hs, (ulong) heap_status)); return(NULL);
} elseif (hs == heap_status) { /* A new record was allocated from the heap. */ if (UNIV_UNLIKELY(val & 1)) { /* Only existing records may be cleared. */
page_zip_fail(("page_zip_apply_log:" " attempting to create" " deleted rec %lu\n",
(ulong) hs)); return(NULL);
}
heap_status += 1 << REC_HEAP_NO_SHIFT;
}
mach_write_to_2(rec - REC_NEW_HEAP_NO, hs);
if (val & 1) { /* Clear the data bytes of the record. */
mem_heap_t* heap = NULL;
rec_offs* offs;
offs = rec_get_offsets(rec, index, offsets, n_core,
ULINT_UNDEFINED, &heap);
memset(rec, 0, rec_offs_data_size(offs));
if (UNIV_LIKELY_NULL(heap)) {
mem_heap_free(heap);
} continue;
}
compile_time_assert(REC_STATUS_NODE_PTR == TRUE);
rec_get_offsets_reverse(data, index,
hs & REC_STATUS_NODE_PTR,
offsets); /* Silence a debug assertion in rec_offs_make_valid(). Thiswillbeoverwritteninpage_zip_set_extra_bytes(),
called by page_zip_decompress_low(). */
ut_d(rec[-REC_NEW_INFO_BITS] = 0);
rec_offs_make_valid(rec, index, n_core != 0, offsets);
/* Copy the extra bytes (backwards). */
{
byte* start = rec_get_start(rec, offsets);
byte* b = rec - REC_N_NEW_EXTRA_BYTES; while (b != start) {
*--b = *data++;
}
}
/* Copy the data bytes. */ if (UNIV_UNLIKELY(rec_offs_any_extern(offsets))) { /* Non-leaf nodes should not contain any
externally stored columns. */ if (UNIV_UNLIKELY(hs & REC_STATUS_NODE_PTR)) {
page_zip_fail(("page_zip_apply_log:" " %lu&REC_STATUS_NODE_PTR\n",
(ulong) hs)); return(NULL);
}
data = page_zip_apply_log_ext(
rec, offsets, trx_id_col, data, end);
if (UNIV_UNLIKELY(!data)) { return(NULL);
}
} elseif (UNIV_UNLIKELY(hs & REC_STATUS_NODE_PTR)) {
len = rec_offs_data_size(offsets)
- REC_NODE_PTR_SIZE; /* Copy the data bytes, except node_ptr. */ if (UNIV_UNLIKELY(data + len >= end)) {
page_zip_fail(("page_zip_apply_log:" " node_ptr %p+%lu >= %p\n",
(constvoid*) data,
(ulong) len,
(constvoid*) end)); return(NULL);
}
memcpy(rec, data, len);
data += len;
} elseif (UNIV_LIKELY(trx_id_col == ULINT_UNDEFINED)) {
len = rec_offs_data_size(offsets);
/* Copy all data bytes of
a record in a secondary index. */ if (UNIV_UNLIKELY(data + len >= end)) {
page_zip_fail(("page_zip_apply_log:" " sec %p+%lu >= %p\n",
(constvoid*) data,
(ulong) len,
(constvoid*) end)); return(NULL);
}
memcpy(rec, data, len);
data += len;
} else { /* Skip DB_TRX_ID and DB_ROLL_PTR. */
ulint l = rec_get_nth_field_offs(offsets,
trx_id_col, &len);
byte* b;
/* Copy any preceding data bytes. */
memcpy(rec, data, l);
data += l;
/* Copy any bytes following DB_TRX_ID, DB_ROLL_PTR. */
b = rec + l + (DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN);
len = ulint(rec_get_end(rec, offsets) - b); if (UNIV_UNLIKELY(data + len >= end)) {
page_zip_fail(("page_zip_apply_log:" " clust %p+%lu >= %p\n",
(constvoid*) data,
(ulong) len,
(constvoid*) end)); return(NULL);
}
memcpy(b, data, len);
data += len;
}
}
}
/**********************************************************************//**
Set the heap_no in a record, and skip the fixed-size record header
that is not included in the d_stream.
@returnTRUE on success, FALSEif d_stream does not end at rec */ static
ibool
page_zip_decompress_heap_no( /*========================*/
z_stream* d_stream, /*!< in/out: compressed page stream */
rec_t* rec, /*!< in/out: record */
ulint& heap_status) /*!< in/out: heap_no and status bits */
{ if (d_stream->next_out != rec - REC_N_NEW_EXTRA_BYTES) { /* n_dense has grown since the page was last compressed. */ return(FALSE);
}
/* Skip the REC_N_NEW_EXTRA_BYTES. */
d_stream->next_out = rec;
/* Set heap_no and the status bits. */
mach_write_to_2(rec - REC_NEW_HEAP_NO, heap_status);
heap_status += 1 << REC_HEAP_NO_SHIFT; return(TRUE);
}
/**********************************************************************//**
Decompress the records of a node pointer page.
@returnTRUE on success, FALSE on failure */ static
ibool
page_zip_decompress_node_ptrs( /*==========================*/
page_zip_des_t* page_zip, /*!< in/out: compressed page */
z_stream* d_stream, /*!< in/out: compressed page stream */
rec_t** recs, /*!< in: dense page directory
sorted by address */
ulint n_dense, /*!< in: size of recs[] */
dict_index_t* index, /*!< in: the index of the page */
rec_offs* offsets, /*!< in/out: temporary offsets */
mem_heap_t* heap) /*!< in: temporary memory heap */
{
ulint heap_status = REC_STATUS_NODE_PTR
| PAGE_HEAP_NO_USER_LOW << REC_HEAP_NO_SHIFT;
ulint slot; const byte* storage;
/* Subtract the space reserved for uncompressed data. */
d_stream->avail_in -= static_cast<uInt>(
n_dense * (PAGE_ZIP_DIR_SLOT_SIZE + REC_NODE_PTR_SIZE));
/* Decompress the records in heap_no order. */ for (slot = 0; slot < n_dense; slot++) {
rec_t* rec = recs[slot];
ut_ad(d_stream->avail_out < srv_page_size
- PAGE_ZIP_START - PAGE_DIR); switch (inflate(d_stream, Z_SYNC_FLUSH)) { case Z_STREAM_END:
page_zip_decompress_heap_no(
d_stream, rec, heap_status); goto zlib_done; case Z_OK: case Z_BUF_ERROR: if (!d_stream->avail_out) { break;
} /* fall through */ default:
page_zip_fail(("page_zip_decompress_node_ptrs:" " 1 inflate(Z_SYNC_FLUSH)=%s\n",
d_stream->msg)); goto zlib_error;
}
if (!page_zip_decompress_heap_no(
d_stream, rec, heap_status)) {
ut_ad(0);
}
/* Read the offsets. The status bits are needed here. */
offsets = rec_get_offsets(rec, index, offsets, 0,
ULINT_UNDEFINED, &heap);
/* Non-leaf nodes should not have any externally
stored columns. */
ut_ad(!rec_offs_any_extern(offsets));
/* Decompress the data bytes, except node_ptr. */
d_stream->avail_out =static_cast<uInt>(
rec_offs_data_size(offsets) - REC_NODE_PTR_SIZE);
switch (inflate(d_stream, Z_SYNC_FLUSH)) { case Z_STREAM_END: goto zlib_done; case Z_OK: case Z_BUF_ERROR: if (!d_stream->avail_out) { break;
} /* fall through */ default:
page_zip_fail(("page_zip_decompress_node_ptrs:" " 2 inflate(Z_SYNC_FLUSH)=%s\n",
d_stream->msg)); goto zlib_error;
}
/* Clear the node pointer in case the record willbedeletedandthespacewillbereallocated
to a smaller record. */
memset(d_stream->next_out, 0, REC_NODE_PTR_SIZE);
d_stream->next_out += REC_NODE_PTR_SIZE;
/* Decompress any trailing garbage, in case the last record was
allocated from an originally longer space on the free list. */
d_stream->avail_out = static_cast<uInt>(
page_header_get_field(page_zip->data, PAGE_HEAP_TOP)
- page_offset(d_stream->next_out)); if (UNIV_UNLIKELY(d_stream->avail_out > srv_page_size
- PAGE_ZIP_START - PAGE_DIR)) {
/* Note that d_stream->avail_out > 0 may hold here
if the modification log is nonempty. */
zlib_done: if (UNIV_UNLIKELY(inflateEnd(d_stream) != Z_OK)) {
ut_error;
}
{
page_t* page = page_align(d_stream->next_out);
/* Clear the unused heap space on the uncompressed page. */
memset(d_stream->next_out, 0,
ulint(page_dir_get_nth_slot(page,
page_dir_get_n_slots(page)
- 1U)
- d_stream->next_out));
}
/* Decompress everything up to this record. */
d_stream->avail_out = static_cast<uint>(
rec - REC_N_NEW_EXTRA_BYTES - d_stream->next_out);
if (UNIV_LIKELY(d_stream->avail_out)) { switch (inflate(d_stream, Z_SYNC_FLUSH)) { case Z_STREAM_END:
page_zip_decompress_heap_no(
d_stream, rec, heap_status); goto zlib_done; case Z_OK: case Z_BUF_ERROR: if (!d_stream->avail_out) { break;
} /* fall through */ default:
page_zip_fail(("page_zip_decompress_sec:" " inflate(Z_SYNC_FLUSH)=%s\n",
d_stream->msg)); goto zlib_error;
}
}
if (!page_zip_decompress_heap_no(
d_stream, rec, heap_status)) {
ut_ad(0);
}
}
/* Decompress the data of the last record and any trailing garbage, incasethelastrecordwasallocatedfromanoriginallylongerspace
on the free list. */
d_stream->avail_out = static_cast<uInt>(
page_header_get_field(page_zip->data, PAGE_HEAP_TOP)
- page_offset(d_stream->next_out)); if (UNIV_UNLIKELY(d_stream->avail_out > srv_page_size
- PAGE_ZIP_START - PAGE_DIR)) {
/* Note that d_stream->avail_out > 0 may hold here
if the modification log is nonempty. */
zlib_done: if (UNIV_UNLIKELY(inflateEnd(d_stream) != Z_OK)) {
ut_error;
}
{
page_t* page = page_align(d_stream->next_out);
/* Clear the unused heap space on the uncompressed page. */
memset(d_stream->next_out, 0,
ulint(page_dir_get_nth_slot(page,
page_dir_get_n_slots(page)
- 1U)
- d_stream->next_out));
}
/* There are no uncompressed columns on leaf pages of
secondary indexes. */
return(TRUE);
}
/**********************************************************************//**
Decompress a record of a leaf node of a clustered index that contains
externally stored columns.
@returnTRUE on success */ static
ibool
page_zip_decompress_clust_ext( /*==========================*/
z_stream* d_stream, /*!< in/out: compressed page stream */
rec_t* rec, /*!< in/out: record */ const rec_offs* offsets, /*!< in: rec_get_offsets(rec) */
ulint trx_id_col) /*!< in: position of of DB_TRX_ID */
{
ulint i;
for (i = 0; i < rec_offs_n_fields(offsets); i++) {
ulint len;
byte* dst;
if (UNIV_UNLIKELY(i == trx_id_col)) { /* Skip trx_id and roll_ptr */
dst = rec_get_nth_field(rec, offsets, i, &len); if (UNIV_UNLIKELY(len < DATA_TRX_ID_LEN
+ DATA_ROLL_PTR_LEN)) {
switch (inflate(d_stream, Z_SYNC_FLUSH)) { case Z_STREAM_END: case Z_OK: case Z_BUF_ERROR: if (!d_stream->avail_out) { break;
} /* fall through */ default:
page_zip_fail(("page_zip_decompress_clust_ext:" " 1 inflate(Z_SYNC_FLUSH)=%s\n",
d_stream->msg)); return(FALSE);
}
ut_ad(d_stream->next_out == dst);
/* Clear DB_TRX_ID and DB_ROLL_PTR in order to avoiduninitializedbytesincasetherecord
is affected by page_zip_apply_log(). */
memset(dst, 0, DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN);
switch (inflate(d_stream, Z_SYNC_FLUSH)) { case Z_STREAM_END: case Z_OK: case Z_BUF_ERROR: if (!d_stream->avail_out) { break;
} /* fall through */ default:
page_zip_fail(("page_zip_decompress_clust:" " 2 inflate(Z_SYNC_FLUSH)=%s\n",
d_stream->msg)); goto zlib_error;
}
ut_ad(d_stream->next_out == dst);
/* Clear DB_TRX_ID and DB_ROLL_PTR in order to avoiduninitializedbytesincasetherecord
is affected by page_zip_apply_log(). */
memset(dst, 0, DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN);
/* Decompress the last bytes of the record. */
d_stream->avail_out = static_cast<uInt>(
rec_get_end(rec, offsets) - d_stream->next_out);
switch (inflate(d_stream, Z_SYNC_FLUSH)) { case Z_STREAM_END: case Z_OK: case Z_BUF_ERROR: if (!d_stream->avail_out) { break;
} /* fall through */ default:
page_zip_fail(("page_zip_decompress_clust:" " 3 inflate(Z_SYNC_FLUSH)=%s\n",
d_stream->msg)); goto zlib_error;
}
}
/* Decompress any trailing garbage, in case the last record was
allocated from an originally longer space on the free list. */
d_stream->avail_out = static_cast<uInt>(
page_header_get_field(page_zip->data, PAGE_HEAP_TOP)
- page_offset(d_stream->next_out)); if (UNIV_UNLIKELY(d_stream->avail_out > srv_page_size
- PAGE_ZIP_START - PAGE_DIR)) {
/* Note that d_stream->avail_out > 0 may hold here
if the modification log is nonempty. */
zlib_done: if (UNIV_UNLIKELY(inflateEnd(d_stream) != Z_OK)) {
ut_error;
}
{
page_t* page = page_align(d_stream->next_out);
/* Clear the unused heap space on the uncompressed page. */
memset(d_stream->next_out, 0,
ulint(page_dir_get_nth_slot(page,
page_dir_get_n_slots(page)
- 1U)
- d_stream->next_out));
}
/* Check if there are any externally stored columnsinthisrecord.Foreachexternally storedcolumn,restoreorclearthe
BTR_EXTERN_FIELD_REF. */ if (!rec_offs_any_extern(offsets)) { continue;
}
for (i = 0; i < rec_offs_n_fields(offsets); i++) { if (!rec_offs_nth_extern(offsets, i)) { continue;
}
dst = rec_get_nth_field(rec, offsets, i, &len);
/**********************************************************************//**
Decompress a page. This function should tolerate errors on the compressed
page. Instead of letting assertions fail, it will returnFALSEif an
inconsistency is detected.
@returnTRUE on success, FALSE on failure */ static
ibool
page_zip_decompress_low( /*====================*/
page_zip_des_t* page_zip,/*!< in: data, ssize;
out: m_start, m_end, m_nonempty, n_blobs */
page_t* page, /*!< out: uncompressed page, may be trashed */
ibool all) /*!< in: TRUE=decompress the whole page; FALSE=verifybutdonotcopysome pageheaderfieldsthatshouldnotchange
after page creation */
{
z_stream d_stream;
dict_index_t* index = NULL;
rec_t** recs; /*!< dense page directory, sorted by address */
ulint n_dense;/* number of user records on the page */
ulint trx_id_col = ULINT_UNDEFINED;
mem_heap_t* heap;
rec_offs* offsets;
d_stream.next_in = page_zip->data + PAGE_DATA; /* Subtract the space reserved for
the page header and the end marker of the modification log. */
d_stream.avail_in = static_cast<uInt>(
page_zip_get_size(page_zip) - (PAGE_DATA + 1));
d_stream.next_out = page + PAGE_ZIP_START;
d_stream.avail_out = uInt(srv_page_size - PAGE_ZIP_START);
if (UNIV_UNLIKELY(inflateInit2(&d_stream, int(srv_page_size_shift))
!= Z_OK)) {
ut_error;
}
/* Decode the zlib header and the index information. */ if (UNIV_UNLIKELY(inflate(&d_stream, Z_BLOCK) != Z_OK)) {
/**********************************************************************//**
Decompress a page. This function should tolerate errors on the compressed
page. Instead of letting assertions fail, it will returnFALSEif an
inconsistency is detected.
@returnTRUE on success, FALSE on failure */
ibool
page_zip_decompress( /*================*/
page_zip_des_t* page_zip,/*!< in: data, ssize;
out: m_start, m_end, m_nonempty, n_blobs */
page_t* page, /*!< out: uncompressed page, may be trashed */
ibool all) /*!< in: TRUE=decompress the whole page; FALSE=verifybutdonotcopysome pageheaderfieldsthatshouldnotchange
after page creation */
{ const ulonglong ns = my_interval_timer();
if (!page_zip_decompress_low(page_zip, page, all)) { return(FALSE);
}
if (srv_cmp_per_index_enabled) {
mysql_mutex_lock(&page_zip_stat_per_index_mutex);
page_zip_stat_per_index[index_id].decompressed++;
page_zip_stat_per_index[index_id].decompressed_usec += time_diff;
mysql_mutex_unlock(&page_zip_stat_per_index_mutex);
}
/* Update the stat counter for LRU policy. */
buf_LRU_stat_inc_unzip();
MONITOR_INC(MONITOR_PAGE_DECOMPRESS);
return(TRUE);
}
#ifdef UNIV_ZIP_DEBUG /**********************************************************************//**
Dump a block of memory on the standard error stream. */ static void
page_zip_hexdump_func( /*==================*/ constchar* name, /*!< in: name of the data structure */ constvoid* buf, /*!< in: data */
ulint size) /*!< in: length of the data, in bytes */
{ const byte* s = static_cast<const byte*>(buf);
ulint addr; const ulint width = 32; /* bytes per line */
/** Dump a block of memory on the standard error stream. @parambufin:data
@param size in: length of the data, in bytes */ #define page_zip_hexdump(buf, size) page_zip_hexdump_func(#buf, buf, size)
/** Flag: make page_zip_validate() compare page headers only */ bool page_zip_validate_header_only;
/**********************************************************************//**
Check that the compressed and decompressed pages match.
@returnTRUEif valid, FALSEifnot */
ibool
page_zip_validate_low( /*==================*/ const page_zip_des_t* page_zip,/*!< in: compressed page */ const page_t* page, /*!< in: uncompressed page */ const dict_index_t* index, /*!< in: index of the page, if known */
ibool sloppy) /*!< in: FALSE=strict,
TRUE=ignore the MIN_REC_FLAG */
{
ibool valid;
/* In crash recovery, the "minimum record" flag may be setincorrectlyuntilthemini-transactionis committed.Letustoleratethatdifferencewhenwe
are performing a sloppy validation. */
/* Only the minimum record flag
differed. Let us ignore it. */
page_zip_fail(("page_zip_validate:" " min_rec_flag" " (%s" UINT32PF "," UINT32PF ",0x%02x)\n",
sloppy ? "ignored, " : "",
page_get_space_id(page),
page_get_page_no(page),
page[offset])); /* We don't check for spatial index, since the"minimumrecord"couldbedeletedwhen doingrtr_update_mbr_field. GIS_FIXME:needtovalidatewhy
rtr_update_mbr_field.() could affect this */ if (index && dict_index_is_spatial(index)) {
valid = true;
} else {
valid = sloppy;
} goto func_exit;
}
}
/* Compare the pointers in the PAGE_FREE list. */
rec = page_header_get_ptr(page, PAGE_FREE);
trec = page_header_get_ptr(temp_page, PAGE_FREE);
/* Note that this will not take into account
the BLOB columns of rec if create==TRUE. */
ut_ad(data + rec_offs_data_size(offsets)
- (DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN)
- n_ext * FIELD_REF_SIZE
< externs - FIELD_REF_SIZE * page_zip->n_blobs);
src += sys_len;
mtr->zmemcpy(*block, sys - page_zip->data,
sys_len); /* Log the last bytes of the record. */
len = rec_offs_data_size(offsets)
- ulint(src - rec);
ASSERT_ZERO(data, len);
memcpy(data, src, len);
data += len;
}
} else { /* Leaf page of a secondary index:
no externally stored columns */
ut_ad(!rec_offs_any_extern(offsets));
/* Log the entire record. */
ulint len = rec_offs_data_size(offsets);
ASSERT_ZERO(data, len);
memcpy(data, rec, len);
data += len;
}
} else { /* This is a node pointer page. */ /* Non-leaf nodes should not have any externally
stored columns. */
ut_ad(!rec_offs_any_extern(offsets));
/* Copy the data bytes, except node_ptr. */
ulint len = rec_offs_data_size(offsets) - REC_NODE_PTR_SIZE;
ut_ad(data + len < storage - REC_NODE_PTR_SIZE
* (page_dir_get_n_heap(page) - PAGE_HEAP_NO_USER_LOW));
ASSERT_ZERO(data, len);
memcpy(data, rec, len);
data += len;
/* Copy the node pointer to the uncompressed area. */
byte* node_ptr = storage - REC_NODE_PTR_SIZE * (heap_no - 1);
mtr->zmemcpy<mtr_t::MAYBE_NOP>(*block, node_ptr,
rec + len, REC_NODE_PTR_SIZE);
}
/**********************************************************************//**
Write a BLOB pointer of a record on the leaf page of a clustered index.
The information must already have been updated on the uncompressed page. */ void
page_zip_write_blob_ptr( /*====================*/
buf_block_t* block, /*!< in/out: ROW_FORMAT=COMPRESSED page */ const byte* rec, /*!< in/out: record whose data is being
written */
dict_index_t* index, /*!< in: index of the page */ const rec_offs* offsets,/*!< in: rec_get_offsets(rec, index) */
ulint n, /*!< in: column index */
mtr_t* mtr) /*!< in/out: mini-transaction */
{ const byte* field;
byte* externs; const page_t* const page = block->page.frame;
page_zip_des_t* const page_zip = &block->page.zip;
ulint blob_no;
ulint len;
/**********************************************************************//**
Clear an area on the uncompressed and compressed page. Donot clear the data payload, as that would grow the modification log. */ static void
page_zip_clear_rec( /*===============*/
buf_block_t* block, /*!< in/out: compressed page */
byte* rec, /*!< in: record to clear */ const dict_index_t* index, /*!< in: index of rec */ const rec_offs* offsets, /*!< in: rec_get_offsets(rec, index) */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
ulint heap_no;
byte* storage;
byte* field;
ulint len;
/* page_zip_validate() would fail here if a record
containing externally stored columns is being deleted. */
ut_ad(rec_offs_validate(rec, index, offsets));
ut_ad(!page_zip_dir_find(page_zip, page_offset(rec)));
ut_ad(page_zip_dir_find_free(page_zip, page_offset(rec)));
ut_ad(page_zip_header_cmp(page_zip, block->page.frame));
if (!page_is_leaf(block->page.frame)) { /* Clear node_ptr. On the compressed page, thereisanarrayofnode_ptrimmediatelybeforethe
dense page directory, at the very end of the page. */
storage = page_zip_dir_start(page_zip);
ut_ad(dict_index_get_n_unique_in_tree_nonleaf(index) ==
rec_offs_n_fields(offsets) - 1);
field = rec_get_nth_field(rec, offsets,
rec_offs_n_fields(offsets) - 1,
&len);
ut_ad(len == REC_NODE_PTR_SIZE);
ut_ad(!rec_offs_any_extern(offsets));
memset(field, 0, REC_NODE_PTR_SIZE);
storage -= (heap_no - 1) * REC_NODE_PTR_SIZE;
len = REC_NODE_PTR_SIZE;
clear_page_zip:
memset(storage, 0, len);
mtr->memset(*block, storage - page_zip->data, len, 0);
} elseif (index->is_clust()) { /* Clear trx_id and roll_ptr. On the compressed page, thereisanarrayofthesefieldsimmediatelybeforethe
dense page directory, at the very end of the page. */ const ulint trx_id_pos
= dict_col_get_clust_pos(
dict_table_get_sys_col(
index->table, DATA_TRX_ID), index);
field = rec_get_nth_field(rec, offsets, trx_id_pos, &len);
ut_ad(len == DATA_TRX_ID_LEN);
memset(field, 0, DATA_TRX_ID_LEN + DATA_ROLL_PTR_LEN);
if (rec_offs_any_extern(offsets)) {
ulint i;
for (i = rec_offs_n_fields(offsets); i--; ) { /* Clear all BLOB pointers in order to make
page_zip_validate() pass. */ if (rec_offs_nth_extern(offsets, i)) {
field = rec_get_nth_field(
rec, offsets, i, &len);
ut_ad(len
== BTR_EXTERN_FIELD_REF_SIZE);
memset(field + len
- BTR_EXTERN_FIELD_REF_SIZE, 0, BTR_EXTERN_FIELD_REF_SIZE);
}
}
}
/**********************************************************************//**
Write the "owned" flag of a record on a compressed page. The n_owned field
must already have been written on the uncompressed page. */ void
page_zip_rec_set_owned( /*===================*/
buf_block_t* block, /*!< in/out: ROW_FORMAT=COMPRESSED page */ const byte* rec, /*!< in: record on the uncompressed page */
ulint flag, /*!< in: the owned flag (nonzero=TRUE) */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
ut_ad(page_align(rec) == block->page.frame);
page_zip_des_t *const page_zip= &block->page.zip;
byte *slot= page_zip_dir_find(page_zip, page_offset(rec));
MEM_CHECK_DEFINED(page_zip->data, page_zip_get_size(page_zip));
byte b= *slot; if (flag)
b|= (PAGE_ZIP_DIR_SLOT_OWNED >> 8); else
b&= byte(~(PAGE_ZIP_DIR_SLOT_OWNED >> 8));
mtr->zmemcpy<mtr_t::MAYBE_NOP>(*block, slot, &b, 1);
}
/**********************************************************************//**
Insert a record to the dense page directory. */ void
page_zip_dir_insert( /*================*/
page_cur_t* cursor, /*!< in/out: page cursor */
uint16_t free_rec,/*!< in: record from which rec was
allocated, or 0 */
byte* rec, /*!< in: record to insert */
mtr_t* mtr) /*!< in/out: mini-transaction */
{
ut_ad(page_align(cursor->rec) == cursor->block->page.frame);
ut_ad(page_align(rec) == cursor->block->page.frame);
page_zip_des_t *const page_zip= &cursor->block->page.zip;
if (page_rec_is_infimum(cursor->rec)) { /* Use the first slot. */
slot_rec = page_zip->data + page_zip_get_size(page_zip);
} else {
byte* end = page_zip->data + page_zip_get_size(page_zip);
byte* start = end - page_zip_dir_user_size(page_zip);
if (UNIV_LIKELY(!free_rec)) { /* PAGE_N_RECS was already incremented inpage_cur_insert_rec_zip(),butthe densedirectoryslotatthatposition
contains garbage. Skip it. */
start += PAGE_ZIP_DIR_SLOT_SIZE;
}
/* Read the old n_dense (n_heap may have been incremented). */
n_dense = page_dir_get_n_heap(page_zip->data)
- (PAGE_HEAP_NO_USER_LOW + 1U);
if (UNIV_UNLIKELY(free_rec)) { /* The record was allocated from the free list. Shiftthedensedirectoryonlyuptothatslot. Notethatinthiscase,n_denseisactually offbyone,becausepage_cur_insert_rec_zip()
did not increment n_heap. */
ut_ad(rec_get_heap_no_new(rec) < n_dense + 1
+ PAGE_HEAP_NO_USER_LOW);
ut_ad(page_offset(rec) >= free_rec);
slot_free = page_zip_dir_find(page_zip, free_rec);
ut_ad(slot_free);
slot_free += PAGE_ZIP_DIR_SLOT_SIZE;
} else { /* The record was allocated from the heap.
Shift the entire dense directory. */
ut_ad(rec_get_heap_no_new(rec) == n_dense
+ PAGE_HEAP_NO_USER_LOW);
/* Shift to the end of the dense page directory. */
slot_free = page_zip->data + page_zip_get_size(page_zip)
- PAGE_ZIP_DIR_SLOT_SIZE * n_dense;
}
if (const ulint slot_len = ulint(slot_rec - slot_free)) { /* Shift the dense directory to allocate place for rec. */
memmove_aligned<2>(slot_free - PAGE_ZIP_DIR_SLOT_SIZE,
slot_free, slot_len);
mtr->memmove(*cursor->block, (slot_free - page_zip->data)
- PAGE_ZIP_DIR_SLOT_SIZE,
slot_free - page_zip->data, slot_len);
}
/* Write the entry for the inserted record.
The "owned" flag must be zero. */
uint16_t offs = page_offset(rec); if (rec_get_deleted_flag(rec, true)) {
offs |= PAGE_ZIP_DIR_SLOT_DEL;
}
if (UNIV_UNLIKELY(!free)) /* Make the last slot the start of the free list. */
slot_free= page_zip->data + page_zip_get_size(page_zip) -
PAGE_ZIP_DIR_SLOT_SIZE * (page_dir_get_n_heap(page_zip->data) -
PAGE_HEAP_NO_USER_LOW); else
{
slot_free= page_zip_dir_find_free(page_zip, page_offset(free));
ut_a(slot_free < slot_rec); /* Grow the free list by one slot by moving the start. */
slot_free+= PAGE_ZIP_DIR_SLOT_SIZE;
}
/* Write the entry for the deleted record.
The "owned" and "deleted" flags will be cleared. */
mach_write_to_2(slot_free, page_offset(rec));
mtr->zmemcpy(*block, slot_free - page_zip->data, 2);
if (const ulint n_ext= rec_offs_n_extern(offsets))
{
ut_ad(index->is_primary());
ut_ad(page_is_leaf(block->page.frame));
/* Shift and zero fill the array of BLOB pointers. */
ulint blob_no = page_zip_get_n_prev_extern(page_zip, rec, index);
ut_a(blob_no + n_ext <= page_zip->n_blobs);
/**********************************************************************//**
Reorganize and compress a page. This is a low-level operation for
compressed pages, to be used when page_zip_compress() fails.
On success, redo log will be written.
The function btr_page_reorganize() should be preferred whenever possible.
@return error code
@retval DB_FAIL on overflow; the block_zip will be left intact */
dberr_t
page_zip_reorganize(
buf_block_t* block, /*!< in/out: page with compressed page; onthecompressedpage,in:size; out:data,n_blobs,
m_start, m_end, m_nonempty */
dict_index_t* index, /*!< in: index of the B-tree node */
ulint z_level,/*!< in: compression level */
mtr_t* mtr, /*!< in: mini-transaction */ bool restore)/*!< whether to restore on failure */
{
page_t* page = buf_block_get_frame(block);
buf_block_t* temp_block;
page_t* temp_page;
ut_ad(mtr->memo_contains_flagged(block, MTR_MEMO_PAGE_X_FIX));
ut_ad(block->page.zip.data);
ut_ad(page_is_comp(page));
ut_ad(!index->table->is_temporary()); /* Note that page_zip_validate(page_zip, page, index) may fail here. */
MEM_CHECK_DEFINED(page, srv_page_size);
MEM_CHECK_DEFINED(buf_block_get_page_zip(block)->data,
page_zip_get_size(buf_block_get_page_zip(block)));
/* Copy the PAGE_MAX_TRX_ID or PAGE_ROOT_AUTO_INC. */
memcpy_aligned<8>(page + (PAGE_HEADER + PAGE_MAX_TRX_ID),
temp_page + (PAGE_HEADER + PAGE_MAX_TRX_ID), 8); /* PAGE_MAX_TRX_ID must be set on secondary index leaf pages. */
ut_ad(err != DB_SUCCESS
|| index->is_clust() || !page_is_leaf(temp_page)
|| page_get_max_trx_id(page) != 0); /* PAGE_MAX_TRX_ID must be zero on non-leaf pages other than
clustered index root pages. */
ut_ad(err != DB_SUCCESS
|| page_get_max_trx_id(page) == 0
|| (index->is_clust()
? !page_has_siblings(temp_page)
: page_is_leaf(temp_page)));
/**********************************************************************//**
Copy the records of a page byte for byte. Donot copy the page header or trailer, except those B-tree header fields that are directly
related to the storage of records. Also copy PAGE_MAX_TRX_ID.
NOTE: The caller must update the lock table and the adaptive hash index. */ void
page_zip_copy_recs(
buf_block_t* block, /*!< in/out: buffer block */ const page_zip_des_t* src_zip, /*!< in: compressed page */ const page_t* src, /*!< in: page */
dict_index_t* index, /*!< in: index of the B-tree */
mtr_t* mtr) /*!< in: mini-transaction */
{
page_t* page = block->page.frame;
page_zip_des_t* page_zip = &block->page.zip;
ut_ad(mtr->memo_contains_flagged(block, MTR_MEMO_PAGE_X_FIX));
ut_ad(mtr->memo_contains_page_flagged(src, MTR_MEMO_PAGE_X_FIX));
ut_ad(!index->table->is_temporary()); #ifdef UNIV_ZIP_DEBUG /* The B-tree operations that call this function may set FIL_PAGE_PREVorPAGE_LEVEL,causingatemporarymin_rec_flag mismatch.Astrictpage_zip_validate()willbeexecutedlater
during the B-tree operations. */
ut_a(page_zip_validate_low(src_zip, src, index, TRUE)); #endif/* UNIV_ZIP_DEBUG */
ut_a(page_zip_get_size(page_zip) == page_zip_get_size(src_zip)); if (UNIV_UNLIKELY(src_zip->n_blobs)) {
ut_a(page_is_leaf(src));
ut_a(dict_index_is_clust(index));
}
/* Copy those B-tree page header fields that are related to therecordsstoredinthepage.Alsocopythefield PAGE_MAX_TRX_ID.Skiptherestofthepageheaderand
trailer. On the compressed page, there is no trailer. */
compile_time_assert(PAGE_MAX_TRX_ID + 8 == PAGE_HEADER_PRIV_END);
memcpy_aligned<2>(PAGE_HEADER + page, PAGE_HEADER + src,
PAGE_HEADER_PRIV_END);
memcpy_aligned<2>(PAGE_DATA + page, PAGE_DATA + src,
srv_page_size - (PAGE_DATA + FIL_PAGE_DATA_END));
memcpy_aligned<2>(PAGE_HEADER + page_zip->data,
PAGE_HEADER + src_zip->data,
PAGE_HEADER_PRIV_END);
memcpy_aligned<2>(PAGE_DATA + page_zip->data,
PAGE_DATA + src_zip->data,
page_zip_get_size(page_zip) - PAGE_DATA);
if (dict_index_is_clust(index)) { /* Reset the PAGE_ROOT_AUTO_INC field when copying
from a root page. */
memset_aligned<8>(PAGE_HEADER + PAGE_ROOT_AUTO_INC
+ page, 0, 8);
memset_aligned<8>(PAGE_HEADER + PAGE_ROOT_AUTO_INC
+ page_zip->data, 0, 8);
} else { /* The PAGE_MAX_TRX_ID must be nonzero on leaf pages
of secondary indexes, and 0 on others. */
ut_ad(!page_is_leaf(src) == !page_get_max_trx_id(src));
}
/* Copy all fields of src_zip to page_zip, except the pointer
to the compressed data page. */
{
page_zip_t* data = page_zip->data; new (page_zip) page_zip_des_t(*src_zip, false);
page_zip->data = data;
}
ut_ad(page_zip_get_trailer_len(page_zip, dict_index_is_clust(index))
+ page_zip->m_end < page_zip_get_size(page_zip));
if (!page_is_leaf(src)
&& UNIV_UNLIKELY(!page_has_prev(src))
&& UNIV_LIKELY(page_has_prev(page))) { /* Clear the REC_INFO_MIN_REC_FLAG of the first user record. */
ulint offs = rec_get_next_offs(page + PAGE_NEW_INFIMUM, TRUE); if (UNIV_LIKELY(offs != PAGE_NEW_SUPREMUM)) {
rec_t* rec = page + offs;
ut_a(rec[-REC_N_NEW_EXTRA_BYTES]
& REC_INFO_MIN_REC_FLAG);
rec[-REC_N_NEW_EXTRA_BYTES]
&= byte(~REC_INFO_MIN_REC_FLAG);
}
}
¤ Diese beiden folgenden Angebotsgruppen bietet das Unternehmen0.129Angebot
(Wie Sie bei der Firma Beratungs- und Dienstleistungen beauftragen können 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.