/**************************************************//**
@file fts/fts0fts.cc
Full Text Search interface
***********************************************************************/
/** Verify if a aux table name is a obsolete table
by looking up the key word in the obsolete table names */ #define FTS_IS_OBSOLETE_AUX_TABLE(table_name) \
(strstr((table_name), "DOC_ID") != NULL \
|| strstr((table_name), "ADDED") != NULL \
|| strstr((table_name), "STOPWORDS") != NULL)
/** This is maximum FTS cache for each table and would be
a configurable variable */
Atomic_relaxed<size_t> fts_max_cache_size;
/** Whether the total memory used for FTS cache is exhausted, and we will
need a sync to free some memory */ bool fts_need_sync = false;
/** Variable specifying the total memory allocated for FTS cache */
Atomic_relaxed<size_t> fts_max_total_cache_size;
/** This is FTS result cache limit for each query and would be
a configurable variable */
size_t fts_result_cache_limit;
/** Variable specifying the maximum FTS max token size */
ulong fts_max_token_size;
/** Variable specifying the minimum FTS max token size */
ulong fts_min_token_size;
/** Time to sleep after DEADLOCK error before retrying operation. */ staticconst std::chrono::milliseconds FTS_DEADLOCK_RETRY_WAIT(100);
/** FTS auxiliary table suffixes that are common to all FT indexes. */ constchar* fts_common_tables[] = { "BEING_DELETED", "BEING_DELETED_CACHE", "CONFIG", "DELETED", "DELETED_CACHE",
NULL
};
/** FTS tokenize parameter for plugin parser */ struct fts_tokenize_param_t {
fts_doc_t* result_doc; /*!< Result doc for tokens */
ulint add_pos; /*!< Added position for tokens */
};
/** Run SYNC on the table, i.e., write out data from the cache to the FTSauxiliaryINDEXtableandclearthecacheattheend. @param[in,out]syncsyncstate @param[in]unlock_cachewhetherunlockcachelockwhenwritenode @param[in]waitwhetherwaitwhenasyncisinprogress
@return DB_SUCCESS if all OK */ static
dberr_t
fts_sync(
fts_sync_t* sync, bool unlock_cache, bool wait,
THD* thd);
/****************************************************************//**
Release all resources help by the words rb tree e.g., the node ilist. */ static void
fts_words_free( /*===========*/
ib_rbt_t* words) /*!< in: rb tree of words */
MY_ATTRIBUTE((nonnull));
/*********************************************************************//** This function fetches the document just inserted right before
we commit the transaction, and tokenize the inserted text data and insert into FTS auxiliary table and its cache. */ static void
fts_add_doc_by_id( /*==============*/
fts_trx_table_t*ftt, /*!< in: FTS trx table */
doc_id_t doc_id); /*!< in: doc id */
/** Create the vector of fts_get_doc_t instances. @param[in,out]cacheftscache
@return vector of fts_get_doc_t instances */ static
ib_vector_t*
fts_get_docs_create(
fts_cache_t* cache);
/** Free the FTS cache.
@param[in,out] cache to be freed */ static void
fts_cache_destroy(fts_cache_t* cache)
{
mysql_mutex_destroy(&cache->lock);
mysql_mutex_destroy(&cache->init_lock);
mysql_mutex_destroy(&cache->deleted_lock);
mysql_mutex_destroy(&cache->doc_id_lock);
pthread_cond_destroy(&cache->sync->cond);
if (cache->stopword_info.cached_stopword) {
rbt_free(cache->stopword_info.cached_stopword);
}
if (cache->sync_heap->arg) {
mem_heap_free(static_cast<mem_heap_t*>(cache->sync_heap->arg));
}
mem_heap_free(cache->cache_heap);
}
/** Get a character set based on precise type. @paramprtypeprecisetype
@return the corresponding character set */
CHARSET_INFO*
fts_get_charset(ulint prtype)
{ #ifdef UNIV_DEBUG switch (prtype & DATA_MYSQL_TYPE_MASK) { case MYSQL_TYPE_BIT: case MYSQL_TYPE_STRING: case MYSQL_TYPE_VAR_STRING: case MYSQL_TYPE_TINY_BLOB: case MYSQL_TYPE_MEDIUM_BLOB: case MYSQL_TYPE_BLOB: case MYSQL_TYPE_LONG_BLOB: case MYSQL_TYPE_VARCHAR: break; default:
ut_error;
} #endif/* UNIV_DEBUG */
/** Load user defined stopword from designated user table @paramftsfulltextstructure @paramstopword_tablestopwordtable @paramstopword_infostopwordinformation
@return whether the operation is successful */ bool fts_load_user_stopword(FTSQueryExecutor *executor, fts_t *fts, constchar *stopword_table,
fts_stopword_t *stopword_info) noexcept
{
trx_t* trx= executor->trx(); if (!fts->dict_locked) dict_sys.lock(SRW_LOCK_CALL); /* Validate the user table existence in the right format */ bool ret= false; constchar *row_end;
stopword_info->charset= fts_valid_stopword_table(
stopword_table, &row_end); if (!stopword_info->charset)
{
cleanup: if (!fts->dict_locked) dict_sys.unlock(); return ret;
}
if (!stopword_info->cached_stopword)
{ /* Create the stopword RB tree with the stopword column
charset. All comparison will use this charset */
stopword_info->cached_stopword= rbt_create_arg_cmp( sizeof(fts_tokenizer_word_t), innobase_fts_text_cmp,
(void*)stopword_info->charset);
}
/* Load the stopword table */
dict_table_t* table=
dict_sys.load_table({stopword_table, strlen(stopword_table)}); if (!table) goto cleanup;
/* Use the passed executor's transaction */
trx->op_info= "Load user stopword table into FTS cache";
ib_rbt_t* stop_words= stopword_info->cached_stopword;
ib_alloc_t* allocator= static_cast<ib_alloc_t*>(stopword_info->heap);
mem_heap_t* heap= static_cast<mem_heap_t*>(allocator->arg);
/* Find the field number for 'value' column */
dict_index_t* clust_index= dict_table_get_first_index(table);
ulint value_field_no= ULINT_UNDEFINED; for (ulint i= 0; i < dict_index_get_n_fields(clust_index); i++)
{ const dict_field_t* field= dict_index_get_nth_field(clust_index, i); if (strcmp(field->name, "value") == 0)
{
value_field_no= i; break;
}
} if (value_field_no == ULINT_UNDEFINED)
{
sql_print_error("InnoDB: Could not find 'value' column in " "stopword table %s", stopword_table); goto cleanup;
}
RecordCallback callback(process_stopword,
[](const dtuple_t*, const rec_t*, const dict_index_t*, const rec_offs*) { return RecordCompareAction::PROCESS;
}); /* Read all records from the stopword table */ for (;;)
{
dberr_t error= executor->read_by_index(
table, clust_index, nullptr, PAGE_CUR_G, callback); if (UNIV_LIKELY(error == DB_SUCCESS))
{
stopword_info->status= STOPWORD_USER_TABLE;
ret= true; break;
} else
{ if (error == DB_LOCK_WAIT_TIMEOUT)
{
sql_print_warning("InnoDB: Lock wait timeout reading " "user stopword table. Retrying!");
trx->error_state= DB_SUCCESS;
} else
{
sql_print_error("InnoDB: Error '%s' while reading user " "stopword table.", ut_strerr(error));
ret= false; break;
}
}
} goto cleanup;
}
/******************************************************************//**
Initialize the index cache. */ static void
fts_index_cache_init( /*=================*/
ib_alloc_t* allocator, /*!< in: the allocator to use */
fts_index_cache_t* index_cache) /*!< in: index cache */
{
ulint i;
/** Construct the name of an internal FTS table for the given table. @param[in]fts_tablemetadataonfulltext-indexedtable @param[out]table_nameanameuptoMAX_FULL_NAME_LEN
@param[in] dict_locked whether dict_sys.latch is being held */ void fts_get_table_name(const fts_table_t* fts_table, char* table_name, bool dict_locked)
{ if (!dict_locked) dict_sys.freeze(SRW_LOCK_CALL);
ut_ad(dict_sys.frozen()); /* Include the separator as well. */ const size_t dbname_len= fts_table->table->name.dblen() + 1;
ut_ad(dbname_len > 1);
memcpy(table_name, fts_table->table->name.m_name, dbname_len); if (!dict_locked) dict_sys.unfreeze();
memcpy(table_name += dbname_len, "FTS_", 4);
table_name += 4; int len; switch (fts_table->type)
{ case FTS_COMMON_TABLE:
len= fts_write_object_id(fts_table->table_id, table_name,
FTS_AUX_MIN_TABLE_ID_LENGTH); break;
/* Create the index cache vector that will hold the inverted indexes. */
cache->indexes = ib_vector_create(
cache->self_heap, sizeof(fts_index_cache_t), 2);
/*******************************************************************//**
Check an index is in the table->indexes list
@returnTRUEif it exists */ static
ibool
fts_in_dict_index( /*==============*/
dict_table_t* table, /*!< in: Table */
dict_index_t* index_check) /*!< in: index to be checked */
{
dict_index_t* index;
for (index = dict_table_get_first_index(table);
index != NULL;
index = dict_table_get_next_index(index)) {
if (index == index_check) { return(TRUE);
}
}
return(FALSE);
}
/*******************************************************************//**
Check an index is in the fts->cache->indexes list
@returnTRUEif it exists */ static
ibool
fts_in_index_cache( /*===============*/
dict_table_t* table, /*!< in: Table */
dict_index_t* index) /*!< in: index to be checked */
{
ulint i;
for (i = 0; i < ib_vector_size(table->fts->cache->indexes); i++) {
fts_index_cache_t* index_cache;
if (index_cache->index == index) { return(TRUE);
}
}
return(FALSE);
}
/*******************************************************************//**
Check indexes in the fts->indexes is also present in index cache and
table->indexes list
@returnTRUEif all indexes match */
ibool
fts_check_cached_index( /*===================*/
dict_table_t* table) /*!< in: Table where indexes are dropped */
{
ulint i;
if (!table->fts || !table->fts->cache) { return(TRUE);
}
for (i = 0; i < ib_vector_size(table->fts->indexes); i++) {
dict_index_t* index;
index = static_cast<dict_index_t*>(
ib_vector_getp(table->fts->indexes, i));
if (!fts_in_index_cache(table, index)) { return(FALSE);
}
if (!fts_in_dict_index(table, index)) { return(FALSE);
}
}
return(TRUE);
}
/** Clear all fts resources when there is no internal DOC_ID andtherearenonewftsindextoadd.
@param[in,out] table table where fts is to be freed */ void fts_clear_all(dict_table_t *table)
{ if (DICT_TF2_FLAG_IS_SET(table, DICT_TF2_FTS_HAS_DOC_ID) ||
!table->fts ||
!ib_vector_is_empty(table->fts->indexes)) return;
for (const dict_index_t *index= dict_table_get_first_index(table);
index; index= dict_table_get_next_index(index)) if (index->type & DICT_FTS) return;
/*******************************************************************//**
Drop auxiliary tables related to an FTS index
@return DB_SUCCESS or error number */
dberr_t
fts_drop_index( /*===========*/
dict_table_t* table, /*!< in: Table where indexes are dropped */
dict_index_t* index, /*!< in: Index to be dropped */
trx_t* trx) /*!< in: Transaction for the drop */
{
ib_vector_t* indexes = table->fts->indexes;
dberr_t err = DB_SUCCESS;
if (cache->get_docs) {
fts_reset_get_doc(cache);
}
return(index_cache);
}
/****************************************************************//**
Release all resources help by the words rb tree e.g., the node ilist. */ static void
fts_words_free( /*===========*/
ib_rbt_t* words) /*!< in: rb tree of words */
{ const ib_rbt_node_t* rbt_node;
/* Free the resources held by a word. */ for (rbt_node = rbt_first(words);
rbt_node != NULL;
rbt_node = rbt_first(words)) {
ulint i;
fts_tokenizer_word_t* word;
word = rbt_value(fts_tokenizer_word_t, rbt_node);
/* Free the ilists of this word. */ for (i = 0; i < ib_vector_size(word->nodes); ++i) {
/*********************************************************************//**
Search the index specific cache for a particular FTS index.
@return the index cache else NULL */
UNIV_INLINE
fts_index_cache_t*
fts_get_index_cache( /*================*/
fts_cache_t* cache, /*!< in: cache to search */ const dict_index_t* index) /*!< in: index to search for */
{ #ifdef SAFE_MUTEX
ut_ad(mysql_mutex_is_owner(&cache->lock)
|| mysql_mutex_is_owner(&cache->init_lock)); #endif/* SAFE_MUTEX */
for (ulint i = 0; i < ib_vector_size(cache->indexes); ++i) {
fts_index_cache_t* index_cache;
/**********************************************************************//**
Find an existing word, orifnot found, create one andreturn it.
@return specified word token */ static
fts_tokenizer_word_t*
fts_tokenizer_word_get( /*===================*/
fts_cache_t* cache, /*!< in: cache */
fts_index_cache_t*
index_cache, /*!< in: index cache */
fts_string_t* text) /*!< in: node text */
{
fts_tokenizer_word_t* word;
ib_rbt_bound_t parent;
mysql_mutex_assert_owner(&cache->lock);
/* If it is a stopword, do not index it */ if (!fts_check_token(text,
cache->stopword_info.cached_stopword,
index_cache->charset)) {
return(NULL);
}
/* Check if we found a match, if not then add word to tree. */ if (rbt_search(index_cache->words, &parent, text) != 0) {
mem_heap_t* heap;
fts_tokenizer_word_t new_word;
/* Take into account the RB tree memory use and the vector. */
cache->total_size += sizeof(new_word)
+ sizeof(ib_rbt_node_t)
+ text->f_len
+ (sizeof(fts_node_t) * 4)
+ sizeof(*new_word.nodes);
ut_ad(rbt_validate(index_cache->words));
}
word = rbt_value(fts_tokenizer_word_t, parent.last);
return(word);
}
/**********************************************************************//**
Add the given doc_id/word positions to the given node's ilist. */ void
fts_cache_node_add_positions( /*=========================*/
fts_cache_t* cache, /*!< in: cache */
fts_node_t* node, /*!< in: word node */
doc_id_t doc_id, /*!< in: doc id */
ib_vector_t* positions) /*!< in: fts_token_t::positions */
{
ulint i;
byte* ptr;
byte* ilist;
ulint enc_len;
ulint last_pos;
byte* ptr_start;
doc_id_t doc_id_delta;
#ifdef SAFE_MUTEX if (cache) {
mysql_mutex_assert_owner(&cache->lock);
} #endif/* SAFE_MUTEX */
ut_ad(doc_id >= node->last_doc_id);
/* Calculate the space required to store the ilist. */
doc_id_delta = doc_id - node->last_doc_id;
enc_len = fts_get_encoded_len(doc_id_delta);
last_pos = 0; for (i = 0; i < ib_vector_size(positions); i++) {
ulint pos = *(static_cast<ulint*>(
ib_vector_get(positions, i)));
/* The 0x00 byte at the end of the token positions list. */
enc_len++;
if ((node->ilist_size_alloc - node->ilist_size) >= enc_len) { /* No need to allocate more space, we can fit in the new
data at the end of the old one. */
ilist = NULL;
ptr = node->ilist + node->ilist_size;
} else {
ulint new_size = node->ilist_size + enc_len;
/* Over-reserve space by a fixed size for small lengths and
by 20% for lengths >= 48 bytes. */ if (new_size < 16) {
new_size = 16;
} elseif (new_size < 32) {
new_size = 32;
} elseif (new_size < 48) {
new_size = 48;
} else {
new_size = new_size * 6 / 5;
}
if (ilist) { /* Copy old ilist to the start of the new one and switch the
new one into place in the node. */ if (node->ilist_size > 0) {
memcpy(ilist, node->ilist, node->ilist_size);
ut_free(node->ilist); if (cache) {
cache->total_size -= node->ilist_size;
}
}
node->ilist = ilist;
}
node->ilist_size += enc_len;
if (node->first_doc_id == FTS_NULL_DOC_ID) {
node->first_doc_id = doc_id;
}
node->last_doc_id = doc_id;
++node->doc_count;
}
/**********************************************************************//**
Add document to the cache. */ static void
fts_cache_add_doc( /*==============*/
fts_cache_t* cache, /*!< in: cache */
fts_index_cache_t*
index_cache, /*!< in: index cache */
doc_id_t doc_id, /*!< in: doc id to add */
ib_rbt_t* tokens) /*!< in: document tokens */
{ const ib_rbt_node_t* node;
ulint n_words;
fts_doc_stats_t* doc_stats;
if (!tokens) { return;
}
mysql_mutex_assert_owner(&cache->lock);
n_words = rbt_size(tokens);
for (node = rbt_first(tokens); node; node = rbt_first(tokens)) {
if (doc_id > cache->sync->max_doc_id) {
cache->sync->max_doc_id = doc_id;
}
}
/** Drop a table. @paramtrxtransaction @paramtable_nameFTS_tablename @paramrenamewhethertorenamebeforedropping @returnerrorcode @retvalDB_SUCCESSifthetablewasdropped
@retval DB_FAIL if the table did not exist */ static dberr_t fts_drop_table(trx_t *trx, constchar *table_name, bool rename)
{ if (dict_table_t *table= dict_table_open_on_name(table_name, true,
DICT_ERR_IGNORE_TABLESPACE))
{
table->release(); if (rename)
{
mem_heap_t *heap= mem_heap_create(FN_REFLEN); char *tmp= dict_mem_create_temporary_tablename(heap, table->name.m_name,
table->id);
dberr_t err= row_rename_table_for_mysql(table->name.m_name, tmp, trx,
RENAME_IGNORE_FK);
mem_heap_free(heap); if (err != DB_SUCCESS)
{
ib::error() << "Unable to rename table " << table_name << ": " << err; return err;
}
} if (dberr_t err= trx->drop_table(*table))
{
ib::error() << "Unable to drop table " << table->name << ": " << err; return err;
}
#ifdef UNIV_DEBUG for (auto &p : trx->mod_tables)
{ if (p.first == table)
p.second.set_aux_table();
} #endif/* UNIV_DEBUG */ return DB_SUCCESS;
}
return DB_FAIL;
}
/****************************************************************//**
Rename a single auxiliary table due to database name change.
@return DB_SUCCESS or error code */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
fts_rename_one_aux_table( /*=====================*/ constchar* new_name, /*!< in: new parent tbl name */ constchar* fts_table_old_name, /*!< in: old aux tbl name */
trx_t* trx) /*!< in: transaction */
{ char fts_table_new_name[MAX_TABLE_NAME_LEN];
ulint new_db_name_len = dict_get_db_name_len(new_name);
ulint old_db_name_len = dict_get_db_name_len(fts_table_old_name);
ulint table_new_name_len = strlen(fts_table_old_name)
+ new_db_name_len - old_db_name_len;
/* Check if the new and old database names are the same, if so,
nothing to do */
ut_ad((new_db_name_len != old_db_name_len)
|| strncmp(new_name, fts_table_old_name, old_db_name_len) != 0);
/* Get the database name from "new_name", and table name
from the fts_table_old_name */
strncpy(fts_table_new_name, new_name, new_db_name_len);
strncpy(fts_table_new_name + new_db_name_len,
strchr(fts_table_old_name, '/'),
table_new_name_len - new_db_name_len);
fts_table_new_name[table_new_name_len] = 0;
/****************************************************************//**
Rename auxiliary tables for all fts index for a table. This(rename)
is due to database name change
@return DB_SUCCESS or error code */
dberr_t
fts_rename_aux_tables( /*==================*/
dict_table_t* table, /*!< in: user Table */ constchar* new_name, /*!< in: new table name */
trx_t* trx) /*!< in: transaction */
{
ulint i;
fts_table_t fts_table;
/** This function make sure that table doesn't haveanyotherreferencecount.
@param table_name table name */ staticvoid fts_table_no_ref_count(constchar *table_name)
{
dict_table_t *table= dict_table_open_on_name(
table_name, true, DICT_ERR_IGNORE_TABLESPACE); if (!table) return;
while (table->get_ref_count() > 1)
{
dict_sys.unlock();
std::this_thread::sleep_for(std::chrono::milliseconds(50));
dict_sys.lock(SRW_LOCK_CALL);
}
table->release();
}
/** Stop the purge thread and check n_ref_count of all auxiliary andcommontableassociatedwiththeftstable. @paramtableparentFTStable @paramalready_stoppedTrueindicatespurgethreadswere
already stopped*/ void purge_sys_t::stop_FTS(const dict_table_t &table, bool already_stopped)
{ if (!already_stopped)
purge_sys.stop_FTS();
if (dberr_t err= fts_drop_table(trx, table_name, rename))
{ if (trx->state != TRX_STATE_ACTIVE) return err; /* We only return the status of the last error. */ if (err != DB_FAIL)
error= err;
}
}
return error;
}
/****************************************************************//**
Drops FTS auxiliary tables for an FTS index
@return DB_SUCCESS or error code */
dberr_t fts_drop_index_tables(trx_t *trx, const dict_index_t &index)
{
ulint i;
fts_table_t fts_table;
dberr_t error = DB_SUCCESS;
/* The precise type calculation is as follows: leastsignificantbyte:MySQLtypecode(notapplicableforsyscols) secondleast:DATA_NOT_NULL|DATA_BINARY_TYPE
third least : the MySQL charset-collation code (DATA_MTYPE_MAX) */
if (new_table == NULL) {
error = DB_FAIL; break;
}
mem_heap_empty(heap);
}
mem_heap_free(heap);
return(error);
}
/******************************************************************//**
Calculate the new state of a row given the existing state and a new event.
@returnnew state of row */ static
fts_row_state
fts_trx_row_get_new_state( /*======================*/
fts_row_state old_state, /*!< in: existing state of row */
fts_row_state event) /*!< in: new event */
{ /* The rules for transforming states:
Itiseasilyseenthattheaboverulesdecomposesuchthatwedonot needtostoretherow'sentirehistoryofevents.Instead,wecan storejustonestatefortherowandupdatethatwhennewevents arrive.Thenwecanimplementtheaboverulesasatwo-dimensional look-uptable,andgetcheckingofinvalidcombinations"forfree"
in the process. */
/* The lookup table for transforming states. old_state is the
Y-axis, event is the X-axis. */ staticconst fts_row_state table[4][4] = { /* I M D N */ /* I */ { FTS_INVALID, FTS_INSERT, FTS_NOTHING, FTS_INVALID }, /* M */ { FTS_INVALID, FTS_MODIFY, FTS_DELETE, FTS_INVALID }, /* D */ { FTS_MODIFY, FTS_INVALID, FTS_INVALID, FTS_INVALID }, /* N */ { FTS_INVALID, FTS_INVALID, FTS_INVALID, FTS_INVALID }
};
result = table[(int) old_state][(int) event];
ut_a(result != FTS_INVALID);
return(result);
}
/** Compare two doubly indirected pointers */ staticint fts_ptr2_cmp(constvoid *p1, constvoid *p2)
{ constvoid *a= **static_cast<constvoid*const*const*>(p1); constvoid *b= **static_cast<constvoid*const*const*>(p2); return b > a ? -1 : a > b;
}
/** Compare a singly indirected pointer to a doubly indirected one */ staticint fts_ptr1_ptr2_cmp(constvoid *p1, constvoid *p2)
{ constvoid *a= *static_cast<constvoid*const*>(p1); constvoid *b= **static_cast<constvoid*const*const*>(p2); return b > a ? -1 : a > b;
}
/* Row id found, update state, and if new state is FTS_NOTHING,
we delete the row from our tree. */ if (parent.result == 0) {
fts_trx_row_t* row = rbt_value(fts_trx_row_t, parent.last);
/*********************************************************************//**
Get the next available document id.
@return DB_SUCCESS if OK */
dberr_t
fts_get_next_doc_id( /*================*/ const dict_table_t* table, /*!< in: table */
doc_id_t* doc_id, /*!< out: new document id */
THD* thd) /*!< in: caller's THD */
{
fts_cache_t* cache = table->fts->cache;
/* If the Doc ID system has not yet been initialized, we willconsulttheCONFIGtableandusertabletore-establish
the initial value of the Doc ID */ if (cache->first_doc_id == FTS_NULL_DOC_ID) {
doc_id_t initial_doc_id = 0;
dberr_t err = fts_init_doc_id(table, thd, &initial_doc_id); if (err != DB_SUCCESS) {
*doc_id = 0; return err;
}
}
if (!DICT_TF2_FLAG_IS_SET(table, DICT_TF2_FTS_HAS_DOC_ID)) {
*doc_id = FTS_NULL_DOC_ID; return(DB_SUCCESS);
}
/** Read the synced document id from the fts configuration table @paramexecutorqueryexecutor @paramtableftstable @paramdoc_iddocumentidtoberead
@return DB_SUCCESS in case of success */ static
dberr_t fts_read_synced_doc_id(FTSQueryExecutor *executor, const dict_table_t *table,
doc_id_t *doc_id) noexcept
{
ut_a(table->fts->doc_col != ULINT_UNDEFINED);
executor->trx()->op_info= "reading synced FTS document id";
ConfigReader reader;
*doc_id= 0;
dberr_t error= executor->read_config_with_lock("synced_doc_id", reader); if (error == DB_SUCCESS)
{ char value_buf[FTS_MAX_INT_LEN];
size_t copy_len= std::min(reader.value_span.size(), sizeof(value_buf) - 1);
memcpy(value_buf, reader.value_span.data(), copy_len);
value_buf[copy_len]= '\0'; int n_parsed= sscanf(value_buf, FTS_DOC_ID_FORMAT, doc_id); if (n_parsed != 1) error= DB_ERROR;
executor->release_lock();
} return error;
}
/** This function fetch the Doc ID from CONFIG table, and compare with theDocIDsupplied.AndstorethelargeronetotheCONFIGtable. @paramexecutorqueryexecutor @paramtableftstable @paramcmp_doc_idDocIDtocompare @paramdoc_idlargerdocumentidaftercomparing"cmp_doc_id"to theonestoredinCONFIGtable
@return DB_SUCCESS if OK */ static
dberr_t
fts_cmp_set_sync_doc_id(
FTSQueryExecutor *executor, const dict_table_t *table,
doc_id_t cmp_doc_id,
doc_id_t *doc_id) noexcept
{
ut_ad(!srv_read_only_mode || recv_sys.rpo);
mysql_mutex_lock(&cache->doc_id_lock); /* For each sync operation, we will add next_doc_id by 1,
so to mark a sync operation */ if (cache->next_doc_id < cache->synced_doc_id + 1) {
cache->next_doc_id = cache->synced_doc_id + 1;
}
mysql_mutex_unlock(&cache->doc_id_lock);
/** Update the last document id. This function could create a new transactiontoupdatethelastdocumentid. @paramexecutorqueryexecutor @paramtabletabletobeupdated @paramdoc_idlastdocumentid
@retval DB_SUCCESS if OK */
dberr_t
fts_update_sync_doc_id(FTSQueryExecutor *executor, const dict_table_t *table,
doc_id_t doc_id) noexcept
{ if (srv_read_only_mode) return DB_READ_ONLY; char id[FTS_MAX_ID_LEN];
snprintf(id, sizeof(id), FTS_DOC_ID_FORMAT, doc_id + 1); return executor->update_config_record("synced_doc_id", id);
}
/*********************************************************************//**
Create a new fts_doc_ids_t.
@returnnew fts_doc_ids_t */
fts_doc_ids_t*
fts_doc_ids_create(void) /*====================*/
{
fts_doc_ids_t* fts_doc_ids;
mem_heap_t* heap = mem_heap_create(512);
/*********************************************************************//** Do commit-phase steps necessary for the deletion of a row.
@return DB_SUCCESS or error code */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
fts_delete( /*=======*/
fts_trx_table_t*ftt, /*!< in: FTS trx table */
fts_trx_row_t* row) /*!< in: row */
{
dict_table_t* table = ftt->table;
doc_id_t doc_id = row->doc_id;
trx_t* trx = ftt->fts_trx->trx;
fts_cache_t* cache = table->fts->cache;
/* we do not index Documents whose Doc ID value is 0 */ if (doc_id == FTS_NULL_DOC_ID) {
ut_ad(!DICT_TF2_FLAG_IS_SET(table, DICT_TF2_FTS_HAS_DOC_ID)); return DB_SUCCESS;
}
/* It is possible we update a record that has not yet been sync-ed intocachefromlastcrash(deleteDocwillnotinitializethe sync).AvoidanyaddedcounteraccountinguntiltheFTScache
is re-established and sync-ed */ if (table->fts->added_synced
&& doc_id > cache->synced_doc_id) {
mysql_mutex_lock(&table->fts->cache->deleted_lock);
/* The Doc ID could belong to those left in ADDEDtablefromlastcrash.Soneedtocheck ifitislessthanfirst_doc_idwhenweinitialize
the Doc ID system after reboot */ if (doc_id >= table->fts->cache->first_doc_id
&& table->fts->cache->added > 0) {
--table->fts->cache->added;
}
/* Only if the row was really deleted. */
ut_a(row->state == FTS_DELETE || row->state == FTS_MODIFY);
}
/* Note the deleted document for OPTIMIZE to purge. */
trx->op_info = "adding doc id to FTS DELETED";
FTSQueryExecutor executor(trx, table);
dberr_t error= executor.insert_common_record("DELETED", doc_id);
/* Increment the total deleted count, this is used to calculate the
number of documents indexed. */ if (error == DB_SUCCESS) {
mysql_mutex_lock(&table->fts->cache->deleted_lock);
/* Check whether the index on FTS_DOC_ID is cluster index */
is_id_cluster = (clust_index == fts_id_index);
mtr_start(&mtr);
/* Search based on Doc ID. Here, we'll need to consider the case
when there is no primary index on Doc ID */ constauto n_uniq = table->fts_n_uniq();
tuple = dtuple_create(heap, n_uniq);
dfield = dtuple_get_nth_field(tuple, 0);
dfield->type.mtype = DATA_INT;
dfield->type.prtype = DATA_NOT_NULL | DATA_UNSIGNED | DATA_BINARY_TYPE;
if (need_sync) {
fts_optimize_request_sync_table(table);
}
mtr_start(&mtr);
if (i < num_idx - 1) { if (doc_pcur->restore_position(
BTR_SEARCH_LEAF, &mtr)
!= btr_pcur_t::SAME_ALL) {
ut_ad("invalid state" == 0);
i = num_idx - 1;
}
}
}
fts_doc_free(&doc);
}
if (!is_id_cluster) {
ut_free(doc_pcur->old_rec_buf);
}
}
func_exit:
mtr_commit(&mtr);
ut_free(pcur.old_rec_buf);
mem_heap_free(heap);
}
/*********************************************************************//**
Get maximum Doc ID in a table if index "FTS_DOC_ID_INDEX" exists
@return max Doc ID or0if index "FTS_DOC_ID_INDEX" does not exist */
doc_id_t
fts_get_max_doc_id( /*===============*/
dict_table_t* table) /*!< in: user table */
{
dict_index_t* index;
dict_field_t* dfield MY_ATTRIBUTE((unused)) = NULL;
doc_id_t doc_id = 0;
mtr_t mtr{nullptr};
btr_pcur_t pcur;
index = table->fts_doc_id_index;
if (!index) { return(0);
}
ut_ad(!index->is_instant());
dfield = dict_index_get_nth_field(index, 0);
mtr.start();
/* fetch the largest indexes value */ if (pcur.open_leaf(false, index, BTR_SEARCH_LEAF, &mtr) == DB_SUCCESS
&& !page_is_empty(btr_pcur_get_page(&pcur))) { const rec_t* rec = NULL;
constexpr ulint doc_id_len= 8;
const byte *data = rec + doc_id_len; if (table->versioned_by_id()) { if (0 == memcmp(data, trx_id_max_bytes, sizeof trx_id_max_bytes)) { break;
}
} else { if (IS_MAX_TIMESTAMP(data)) { break;
}
}
} while (btr_pcur_move_to_prev(&pcur, &mtr));
if (!rec || rec_is_metadata(rec, *index)) { goto func_exit;
}
doc_id = fts_read_doc_id(rec);
}
func_exit:
mtr.commit(); return(doc_id);
}
/** Write out a single word's data as new entry/entries in the INDEX table. @paramexecutorFTSQueryExecutor @paramselectedauxiliaryindexnumber @paramaux_dataauxiliarytabledata
@return DB_SUCCESS if all OK or error code */
dberr_t fts_write_node(FTSQueryExecutor *executor, uint8_t selected, const fts_aux_data_t *aux_data) noexcept
{
dberr_t error= executor->insert_aux_record(selected, aux_data); return error;
}
/** Sort an array of doc_id */ void fts_doc_ids_sort(ib_vector_t *doc_ids)
{
doc_id_t *const data= reinterpret_cast<doc_id_t*>(doc_ids->data);
std::sort(data, data + doc_ids->used);
}
/** Add rows to the DELETED_CACHE table.
@return DB_SUCCESS if all went well else error code*/ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
fts_sync_add_deleted_cache(FTSQueryExecutor *executor, fts_sync_t *sync,
ib_vector_t *doc_ids) noexcept
{
ulint n_elems= ib_vector_size(doc_ids);
ut_a(n_elems > 0);
fts_doc_ids_sort(doc_ids);
dberr_t error= DB_SUCCESS; for (ulint i= 0; i < n_elems && error == DB_SUCCESS; ++i)
{
doc_id_t *update= static_cast<doc_id_t*>(ib_vector_get(doc_ids, i));
error= executor->insert_common_record("DELETED_CACHE", *update);
} return error;
}
/** Write the words and ilist to disk @param[in,out]executorqueryexecutor @param[in]index_cacheindexcache @param[in]unlock_cachewhetherunlockcachewhenwritenode
@return DB_SUCCESS if all went well else error code */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t fts_sync_write_words(FTSQueryExecutor *executor,
fts_index_cache_t *index_cache, bool unlock_cache) noexcept
{
dict_table_t *table= index_cache->index->table; bool print_error= false;
dberr_t error= DB_SUCCESS; for (const ib_rbt_node_t *rbt_node= rbt_first(index_cache->words);
rbt_node; rbt_node= rbt_next(index_cache->words, rbt_node))
{
fts_tokenizer_word_t *word= rbt_value(fts_tokenizer_word_t, rbt_node);
DBUG_EXECUTE_IF("fts_instrument_write_words_before_select_index",
std::this_thread::sleep_for(
std::chrono::milliseconds(300)););
uint8_t selected= fts_select_index(
index_cache->charset, word->text.f_str, word->text.f_len);
for (ulint i = 0; i < ib_vector_size(word->nodes); ++i)
{
fts_node_t* fts_node= static_cast<fts_node_t*>(ib_vector_get(word->nodes, i)); if (fts_node->synced) continue; else fts_node->synced= true; /* FIXME: we need to handle the error properly. */ if (error == DB_SUCCESS)
{ if (unlock_cache) mysql_mutex_unlock(&table->fts->cache->lock);
fts_aux_data_t aux_data((constchar*)word->text.f_str, word->text.f_len,
fts_node->first_doc_id, fts_node->last_doc_id, static_cast<uint32_t>(fts_node->doc_count), fts_node->ilist,
fts_node->ilist_size);
error= fts_write_node(executor, selected, &aux_data);
DEBUG_SYNC_C("fts_write_node");
DBUG_EXECUTE_IF("fts_write_node_crash", DBUG_SUICIDE(););
DBUG_EXECUTE_IF("fts_instrument_sync_sleep",
std::this_thread::sleep_for(std::chrono::seconds(1)););
if (unlock_cache) mysql_mutex_lock(&table->fts->cache->lock);
}
if (UNIV_UNLIKELY(error != DB_SUCCESS) && !print_error)
{
sql_print_error("InnoDB: ( %s ) writing word node to FTS auxiliary " "index table %s", ut_strerr(error),
table->name.m_name);
print_error= true;
}
}
}
return error;
}
/*********************************************************************//**
Begin Sync, create transaction, acquire locks, etc. */ static void
fts_sync_begin( /*===========*/
fts_sync_t* sync, /*!< in: sync state */
THD* thd) /*!< in: caller's THD; may be
NULL (e.g. background sync) */
{
sync->trx = trx_create();
sync->trx->mysql_thd = thd;
trx_start_internal(sync->trx);
}
/** Run SYNC on the table, i.e., write out data from the index specificcachetotheFTSauxINDEXtableandFTSauxdocid statstable. @paramexecutorqueryexecutor @paramsyncsyncstate @paramindex_cacheindexcache
@return DB_SUCCESS if all OK */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
fts_sync_index(FTSQueryExecutor *executor, fts_sync_t *sync,
fts_index_cache_t *index_cache) noexcept
{
trx_t *trx= sync->trx;
trx->op_info= "doing SYNC index";
ut_ad(rbt_validate(index_cache->words)); return fts_sync_write_words(executor, index_cache, sync->unlock_cache);
}
/** Check if index cache has been synced completely @param[in,out]index_cacheindexcache
@return true if index is synced, otherwise false. */ static bool
fts_sync_index_check(
fts_index_cache_t* index_cache)
{ const ib_rbt_node_t* rbt_node;
/** Reset synced flag in index cache when rollback
@param[in,out] index_cache index cache */ static void
fts_sync_index_reset(
fts_index_cache_t* index_cache)
{ const ib_rbt_node_t* rbt_node;
/** Commit the SYNC, change state of processed doc ids etc. @param[in,out]executorqueryexecutor @param[in,out]syncsyncstate
@return DB_SUCCESS if all OK */ static MY_ATTRIBUTE((nonnull, warn_unused_result))
dberr_t
fts_sync_commit(FTSQueryExecutor *executor, fts_sync_t *sync) noexcept
{
dberr_t error;
trx_t* trx = sync->trx;
fts_cache_t* cache = sync->table->fts->cache;
doc_id_t last_doc_id;
trx->op_info = "doing SYNC commit";
/* After each Sync, update the CONFIG table about the max doc id
we just sync-ed to index table */
error = fts_cmp_set_sync_doc_id(executor, sync->table, sync->max_doc_id,
&last_doc_id);
/* Get the list of deleted documents that are either in the cacheorwereheadedtherebutweredeletedbeforetheadd
thread got to them. */
if (error == DB_SUCCESS && ib_vector_size(cache->deleted_doc_ids) > 0) {
/* We need to do this within the deleted lock since fts_delete() can
attempt to add a deleted doc id to the cache deleted id array. */
fts_cache_clear(cache);
DEBUG_SYNC_C("fts_deleted_doc_ids_clear");
fts_cache_init(cache);
mysql_mutex_unlock(&cache->lock);
if (UNIV_LIKELY(error == DB_SUCCESS)) {
DEBUG_SYNC_C("fts_crash_before_commit_sync");
fts_sql_commit(trx);
} else {
fts_sql_rollback(trx);
ib::error() << "(" << error << ") during SYNC of " "table " << sync->table->name;
}
/* Avoid assertion in trx_t::free(). */
trx->dict_operation_lock_mode = false;
trx->clear_and_free();
/** Run SYNC on the table, i.e., write out data from the cache to the FTSauxiliaryINDEXtableandclearthecacheattheend. @param[in,out]syncsyncstate @param[in]unlock_cachewhetherunlockcachelockwhenwritenode @param[in]waitwhetherwaitwhenasyncisinprogress
@return DB_SUCCESS if all OK */ static
dberr_t
fts_sync(
fts_sync_t* sync, bool unlock_cache, bool wait,
THD* thd)
{
ut_ad(!srv_read_only_mode || recv_sys.rpo);
if (cache->total_size == 0) {
mysql_mutex_unlock(&cache->lock); return DB_SUCCESS;
}
/* Check if cache is being synced. Note:wereleasecachelockinfts_sync_write_words()to
avoid long wait for the lock by other threads. */ if (sync->in_progress) { if (!wait) {
mysql_mutex_unlock(&cache->lock); return(DB_SUCCESS);
} do {
my_cond_wait(&sync->cond, &cache->lock.m_mutex);
} while (sync->in_progress);
}
const size_t fts_cache_size= fts_max_cache_size;
DEBUG_SYNC_C("fts_sync_begin");
fts_sync_begin(sync, thd);
FTSQueryExecutor executor(sync->trx, sync->table);
error = executor.open_config_table(); if (error) goto end_sync;
begin_sync: if (cache->total_size > fts_cache_size) { /* Avoid the case: sync never finish when
insert/update keeps comming. */
ut_ad(sync->unlock_cache);
sync->unlock_cache = false;
ib::warn() << "Total InnoDB FTS size "
<< cache->total_size << " for the table "
<< cache->sync->table->name
<< " exceeds the innodb_ft_cache_size "
<< fts_cache_size;
}
for (i = 0; i < ib_vector_size(cache->indexes); ++i) {
fts_index_cache_t *index_cache= static_cast<fts_index_cache_t*>(
ib_vector_get(cache->indexes, i));
if (index_cache->index->to_be_dropped) { continue;
}
if (!sync->unlock_cache
&& cache->total_size < fts_max_cache_size) { /* Reset the unlock cache if the value
is less than innodb_ft_cache_size */
sync->unlock_cache = true;
}
}
/* We need to check whether an optimize is required, for that wemakecopiesofthetwovariablesthatcontrolthetrigger.These variablescanchangebehindourbackandwedon'twanttoholdthe
lock for longer than is needed. */
mysql_mutex_lock(&cache->deleted_lock);
cache->added = 0;
cache->deleted = 0;
mysql_mutex_unlock(&cache->deleted_lock);
return(error);
}
/** Run SYNC on the table, i.e., write out data from the cache to the FTSauxiliaryINDEXtableandclearthecacheattheend. @param[in,out]tableftstable @param[in]waitwhetherwaitforexistingsynctofinish
@return DB_SUCCESS on success, error code on failure. */
dberr_t fts_sync_table(dict_table_t* table, bool wait, THD* thd)
{
ut_ad(table->fts);
/** Check if a fts token is a stopword or less than fts_min_token_size orgreaterthanfts_max_token_size. @param[in]tokentokenstring @param[in]stopwordsstopwordsrbtree @param[in]cstokencharset @retvaltrueifitisnotstopwordandlengthinrange
@retval false if it is stopword or length not in range */ bool
fts_check_token( const fts_string_t* token, const ib_rbt_t* stopwords, const CHARSET_INFO* cs)
{
ut_ad(cs != NULL || stopwords == NULL);
/** Add the token and its start position to the token's list of positions. @param[in,out]result_docresultdocrbtree @param[in]strtokenstring
@param[in] position token position */ static void
fts_add_token(
fts_doc_t* result_doc,
fts_string_t str,
ulint position)
{ /* Ignore string whose character number is less than
"fts_min_token_size" or more than "fts_max_token_size" */
if (fts_check_token(&str, NULL, result_doc->charset)) {
/* For binary collations, a case sensitive search is
performed. Hence don't convert to lower case. */ if (my_binary_compare(result_doc->charset)) {
memcpy(t_str.f_str, str.f_str, str.f_len);
t_str.f_str[str.f_len]= 0;
t_str.f_len= str.f_len;
} else {
t_str.f_len= result_doc->charset->casedn_z(
(constchar*) str.f_str, str.f_len,
(char *) t_str.f_str, t_str.f_len);
}
/* Add the word to the document statistics. If the word
hasn't been seen before we create a new entry for it. */ if (rbt_search(result_doc->tokens, &parent, &t_str) != 0) {
fts_token_t new_token;
/******************************************************************** Processnexttokenfromdocumentstartingatthegivenposition,i.e.,add thetoken'sstartpositiontothetoken'slistofpositions.
@return number of characters handled in this call */ static
ulint
fts_process_token( /*==============*/
fts_doc_t* doc, /* in/out: document to
tokenize */
fts_doc_t* result, /* out: if provided, save
result here */
ulint start_pos, /*!< in: start position in text */
ulint add_pos) /*!< in: add this position to all
tokens from this tokenization */
{
ulint ret;
fts_string_t str;
ulint position;
fts_doc_t* result_doc;
byte buf[FTS_MAX_WORD_LEN + 1];
str.f_str = buf;
/* Determine where to save the result. */
result_doc = (result != NULL) ? result : doc;
/* The length of a string in characters is set here only. */
/* const_cast is for reinterpret_cast below, or it will fail. */
start = const_cast<char*>(token);
end = start + len; while (start < end) { int ctype; int mbl;
/*********************************************************************//**
Get the initial Doc ID by consulting the CONFIG table.
On success, *doc_id holds the resolved id and the FTS cache's
first_doc_id is initialized. On failure (e.g. DB_INTERRUPTED when the
caller's thread is killed mid CONFIG-table read), *doc_id is 0 and the
cache is left uninitialized so a later attempt can retry.
@return DB_SUCCESS or error code */
dberr_t
fts_init_doc_id( /*============*/ const dict_table_t* table, /*!< in: table */
THD* thd, /*!< in: caller's THD */
doc_id_t* doc_id) /*!< out: initial Doc ID */
{
*doc_id = 0;
mysql_mutex_lock(&table->fts->cache->lock);
/* Return if the table is already initialized for DOC ID */ if (table->fts->cache->first_doc_id != FTS_NULL_DOC_ID) {
mysql_mutex_unlock(&table->fts->cache->lock); return DB_SUCCESS;
}
DEBUG_SYNC_C("fts_initialize_doc_id");
/* Then compare this value with the ID value stored in the CONFIG
table. The larger one will be our new initial Doc ID */
doc_id_t max_doc_id = 0;
trx_t* trx = trx_create();
trx->mysql_thd = thd;
trx_start_internal_read_only(trx);
FTSQueryExecutor executor(trx, table);
dberr_t err = fts_cmp_set_sync_doc_id(&executor, table, 0,
&max_doc_id);
fts_sql_commit(trx);
trx->free();
if (err != DB_SUCCESS) { /* Leave the cache uninitialized (first_doc_id == FTS_NULL_DOC_ID)soalaterattemptcanretry;doNOT
set added_synced. */
mysql_mutex_unlock(&table->fts->cache->lock); return err;
}
/* If DICT_TF2_FTS_ADD_DOC_ID is set, we are in the process of creatingindex(andadddocidcolumn.Noneedtorecovery
documents */ if (!DICT_TF2_FLAG_IS_SET(table, DICT_TF2_FTS_ADD_DOC_ID)) {
fts_init_index((dict_table_t*) table, TRUE, thd);
}
table->fts->added_synced = true;
table->fts->cache->first_doc_id = max_doc_id;
mysql_mutex_unlock(&table->fts->cache->lock);
ut_ad(max_doc_id > 0);
*doc_id = max_doc_id; return DB_SUCCESS;
}
/*********************************************************************//**
Free the modified rows of a table. */
UNIV_INLINE void
fts_trx_table_rows_free( /*====================*/
ib_rbt_t* rows) /*!< in: rbt of rows to free */
{ const ib_rbt_node_t* node;
/* This can be NULL if a savepoint was released. */ if (ftt->rows != NULL) {
fts_trx_table_rows_free(ftt->rows);
ftt->rows = NULL;
}
/* This can be NULL if a savepoint was released. */ if (ftt->added_doc_ids != NULL) {
fts_doc_ids_free(ftt->added_doc_ids);
ftt->added_doc_ids = NULL;
}
/* The default savepoint name must be NULL. */ if (ftt->docs_added_graph) {
que_graph_free(ftt->docs_added_graph);
}
/* NOTE: We are responsible for free'ing the node */
ut_free(rbt_remove_node(tables, node));
}
/* The default savepoint name must be NULL. */ if (i == 0) {
ut_a(savepoint->name == NULL);
}
fts_savepoint_free(savepoint);
}
if (fts_trx->heap) {
mem_heap_free(fts_trx->heap);
}
}
/*********************************************************************//**
Extract the doc id from the FTS hidden column.
@return doc id that was extracted from rec */
doc_id_t
fts_get_doc_id_from_row( /*====================*/
dict_table_t* table, /*!< in: table */
dtuple_t* row) /*!< in: row whose FTS doc id we
want to extract.*/
{
dfield_t* field;
doc_id_t doc_id = 0;
ut_a(table->fts->doc_col != ULINT_UNDEFINED);
field = dtuple_get_nth_field(row, table->fts->doc_col);
/** Extract the doc id from the record that belongs to index. @param[in]recrecordcontainingFTS_DOC_ID @param[in]indexindexofrec @param[in]offsetsrec_get_offsets(rec,index)
@return doc id that was extracted from rec */
doc_id_t
fts_get_doc_id_from_rec( const rec_t* rec, const dict_index_t* index, const rec_offs* offsets)
{
ulint f = dict_col_get_index_pos(
&index->table->cols[index->table->fts->doc_col], index);
ulint len;
doc_id_t doc_id = mach_read_from_8(
rec_get_nth_field(rec, offsets, f, &len));
ut_ad(len == 8); return doc_id;
}
/*********************************************************************//**
Search the index specific cache for a particular FTS index.
@return the index specific cache else NULL */
fts_index_cache_t*
fts_find_index_cache( /*=================*/ const fts_cache_t* cache, /*!< in: cache to search */ const dict_index_t* index) /*!< in: index to search for */
{ /* We cast away the const because our internal function, takes
non-const cache arg and returns a non-const pointer. */ return(static_cast<fts_index_cache_t*>(
fts_get_index_cache((fts_cache_t*) cache, index)));
}
/*********************************************************************//**
Search cache for word.
@return the word node vector if found else NULL */ const ib_vector_t*
fts_cache_find_word( /*================*/ const fts_index_cache_t*index_cache, /*!< in: cache to search */ const fts_string_t* text) /*!< in: word to search for */
{
ib_rbt_bound_t parent; const ib_vector_t* nodes = NULL;
/* Lookup the word in the rb tree */ if (rbt_search(index_cache->words, &parent, text) == 0) { const fts_tokenizer_word_t* word;
word = rbt_value(fts_tokenizer_word_t, parent.last);
nodes = word->nodes;
}
return(nodes);
}
/*********************************************************************//**
Append deleted doc ids to vector. */ void
fts_cache_append_deleted_doc_ids( /*=============================*/
fts_cache_t* cache, /*!< in: cache to use */
ib_vector_t* vector) /*!< in: append to this vector */
{
mysql_mutex_lock(&cache->deleted_lock);
if (cache->deleted_doc_ids) for (ulint i= 0; i < ib_vector_size(cache->deleted_doc_ids); ++i)
{
doc_id_t *update= static_cast<doc_id_t*>(
ib_vector_get(cache->deleted_doc_ids, i));
ib_vector_push(vector, &update);
}
mysql_mutex_unlock(&cache->deleted_lock);
}
/*********************************************************************//**
Add the FTS document id hidden column. */ void
fts_add_doc_id_column( /*==================*/
dict_table_t* table, /*!< in/out: Table with FTS index */
mem_heap_t* heap) /*!< in: temporary memory heap, or NULL */
{
dict_mem_table_add_col(
table, heap,
FTS_DOC_ID.str,
DATA_INT,
dtype_form_prtype(
DATA_NOT_NULL | DATA_UNSIGNED
| DATA_BINARY_TYPE | DATA_FTS_DOC_ID, 0), sizeof(doc_id_t));
DICT_TF2_FLAG_SET(table, DICT_TF2_FTS_HAS_DOC_ID);
}
/** Add new fts doc id to the update vector. @param[in]tablethetablethatcontainstheFTSindex. @param[in,out]ufieldtheftsdocidfieldintheupdatevector. Nonewmemoryisallocatedforthisinthis function. @param[in,out]next_doc_idtheftsdocidthathasbeenaddedtothe updatevector.If0,anewftsdocidis automaticallygenerated.Thememoryprovided forthisargumentwillbeusedbytheupdate vector.Ensurethatthelifetimeofthis memorymatchesthatoftheupdatevector.
@return the fts doc id used in the update vector */
doc_id_t
fts_update_doc_id(
dict_table_t* table,
upd_field_t* ufield,
doc_id_t* next_doc_id,
THD* thd)
{
doc_id_t doc_id;
dberr_t error = DB_SUCCESS;
if (*next_doc_id) {
doc_id = *next_doc_id;
} else { /* Get the new document id that will be added. */
error = fts_get_next_doc_id(table, &doc_id, thd);
}
if (error == DB_SUCCESS) {
dict_index_t* clust_index;
dict_col_t* col = dict_table_get_nth_col(
table, table->fts->doc_col);
if (last_savepoint->tables != NULL) {
fts_savepoint_copy(last_savepoint, savepoint);
}
}
/*********************************************************************//**
Lookup a savepoint instance.
@return0ifnot found */ static
ulint
fts_savepoint_lookup( /*==================*/
ib_vector_t* savepoints, /*!< in: savepoints */ constvoid* name) /*!< in: savepoint */
{
ut_a(ib_vector_size(savepoints) > 0); for (ulint i= 1; i < ib_vector_size(savepoints); ++i) if (name == static_cast<const fts_savepoint_t*>
(ib_vector_get(savepoints, i))->name) return i; return0;
}
/*********************************************************************//**
Release the savepoint data identified by name. All savepoints created
after the named savepoint are kept.
@return DB_SUCCESS or error code */ void
fts_savepoint_release( /*==================*/
trx_t* trx, /*!< in: transaction */ constvoid* name) /*!< in: savepoint name */
{
ut_a(name != NULL);
if (ulint i = fts_savepoint_lookup(savepoints, name)) {
fts_savepoint_t* savepoint;
savepoint = static_cast<fts_savepoint_t*>(
ib_vector_get(savepoints, i));
if (i == ib_vector_size(savepoints) - 1) { /* If the savepoint is the last, we save its
tables to the previous savepoint. */
fts_savepoint_t* prev_savepoint;
prev_savepoint = static_cast<fts_savepoint_t*>(
ib_vector_get(savepoints, i - 1));
/* We pop all savepoints from the the top of the stack up to
and including the instance that was found. */
ulint i = fts_savepoint_lookup(savepoints, name);
if (i == 0) { /* fts_trx_create() must have been invoked after thissavepointhadbeencreated,andwemustrollback
everything. */
i = 1;
}
{
fts_savepoint_t* savepoint;
while (ib_vector_size(savepoints) > i) {
savepoint = static_cast<fts_savepoint_t*>(
ib_vector_pop(savepoints));
if (savepoint->name != NULL) { /* Since name was allocated on the heap, the memorywillbereleasedwhenthetransaction
completes. */
savepoint->name = NULL;
fts_savepoint_free(savepoint);
}
}
/* Pop all a elements from the top of the stack that may havebeenreleased.Wehavetobecarefulthatwedon't
delete the implied savepoint. */
/* We will start the match after the '/' */
++ptr;
len= end - ptr;
/* All auxiliary tables are prefixed with "FTS_" and the name
length will be at the very least greater than 20 bytes. */ if (len > 24 && !memcmp(ptr, "FTS_", 4))
{ /* Skip the prefix. */
ptr+= 4;
len-= 4;
/* Skip the underscore. */
++ptr;
ut_ad(end > ptr);
len= end - ptr;
sscanf(table_id_ptr, UINT64PFx, table_id); /* First search the common table suffix array. */ for (ulint i = 0; fts_common_tables[i]; ++i)
{ if (!strncmp(ptr, fts_common_tables[i], len)) returntrue;
}
/* Could be obsolete common tables. */ if ((len == 5 && !memcmp(ptr, "ADDED", len)) ||
(len == 9 && !memcmp(ptr, "STOPWORDS", len))) returntrue;
constchar* index_id_ptr= ptr; /* Skip the index id. */
ptr= static_cast<constchar*>(memchr(ptr, '_', len)); if (!ptr) returnfalse;
sscanf(index_id_ptr, UINT64PFx, index_id);
/* Skip the underscore. */
++ptr;
ut_a(end > ptr);
len= end - ptr;
if (len <= 4) returnfalse;
len-= 4; /* .ibd suffix */
if (len > 7) returnfalse;
/* Search the FT index specific array. */ for (ulint i = 0; i < FTS_NUM_AUX_INDEX; ++i)
{ if (!memcmp(ptr, "INDEX_", len - 1)) returntrue;
}
/* Other FT index specific table(s). */ if (len == 6 && !memcmp(ptr, "DOC_ID", len)) returntrue;
}
returnfalse;
}
/**********************************************************************//**
Check whether user supplied stopword table is of the right format.
Caller is responsible to hold dictionary locks.
@param stopword_table_name table name
@param row_end name of the system-versioning end column, or"value"
@return the stopword column charset
@retval NULL if the table does not exist or qualify */
CHARSET_INFO*
fts_valid_stopword_table( /*=====================*/ constchar* stopword_table_name, /*!< in: Stopword table
name */ constchar** row_end) /* row_end value of system-versioned table */
{
dict_table_t* table;
dict_col_t* col = NULL;
if (!table) {
ib::error() << "User stopword table " << stopword_table_name
<< " does not exist.";
return(NULL);
} else { if (strcmp(dict_table_get_col_name(table, 0).str, "value")) {
ib::error() << "Invalid column name for stopword" " table " << stopword_table_name << ". Its" " first column must be named as 'value'.";
return(NULL);
}
col = dict_table_get_nth_col(table, 0);
if (col->mtype != DATA_VARCHAR
&& col->mtype != DATA_VARMYSQL) {
ib::error() << "Invalid column type for stopword" " table " << stopword_table_name << ". Its" " first column must be of varchar type";
/* Load user stopword table if specified, otherwise load default */ if (stopword_to_use
&& fts_load_user_stopword(executor, table->fts, stopword_to_use,
&cache->stopword_info)) { /* Successfully loaded user stopword table */
} else { /* Load system default stopword list */
fts_load_default_stopword(&cache->stopword_info);
}
/* Initialize cached_stopword RB-tree if not already created */ if (!cache->stopword_info.cached_stopword) {
cache->stopword_info.cached_stopword = rbt_create_arg_cmp( sizeof(fts_tokenizer_word_t), innobase_fts_text_cmp,
&my_charset_latin1);
}
returntrue;
}
/** Callback function when we initialize the FTS at the start up time.ItrecoversDocIDsthathavenotsync-edtotheauxiliary table,andrequiretobringthembackintoFTSindex. @paramexecutorqueryexecutor @paramget_docDocument @paramdoc_iddocumentidtobefetched
@return: always returns TRUE */ staticvoid fts_init_recover_all_docs(FTSQueryExecutor *executor,
fts_get_doc_t *get_doc,
doc_id_t doc_id) noexcept
{
executor->trx()->op_info= "fetching indexed FTS document";
dict_index_t *fts_index= get_doc->index_cache->index;
dict_table_t *user_table= fts_index->table;
dict_index_t *fts_doc_id_index= user_table->fts_doc_id_index;
dict_index_t *clust_index= dict_table_get_first_index(user_table);
fts_cache_t *cache= get_doc->cache;
ut_a(user_table->fts->doc_col != ULINT_UNDEFINED);
ut_a(fts_doc_id_index); /* Map FTS index columns to clustered index field positions */
ulint *clust_field_nos= static_cast<ulint*>(
mem_heap_alloc(executor->get_heap(),
fts_index->n_user_defined_cols * sizeof(ulint)));
for (unsigned i= 0; i < fts_index->n_user_defined_cols; i++)
{
dict_field_t* fts_field= dict_index_get_nth_field(fts_index, i);
clust_field_nos[i]= dict_col_get_index_pos(fts_field->col, clust_index);
}
/** Get the next large document id and update it in fulltext cache @paramexecutorqueryexecutor @paramdoc_iddocumentidtobeupdated
@param index fulltext index */ staticvoid fts_init_get_doc_id(FTSQueryExecutor *executor,
doc_id_t doc_id, dict_index_t *index) noexcept
{
executor->trx()->op_info= "fetching indexed FTS document";
dict_table_t* user_table= index->table;
fts_cache_t* cache= user_table->fts->cache;
ut_a(user_table->fts->doc_col != ULINT_UNDEFINED);
/**********************************************************************//** This function brings FTS index in sync when FTS index is first
used. There are documents that have not yet sync-ed to auxiliary
tables from last server abnormally shutdown, we will need to bring
such document into FTS cache before any further operations */ void
fts_init_index( /*===========*/
dict_table_t* table, /*!< in: Table with FTS */ bool has_cache_lock, /*!< in: Whether we already have
cache lock */
THD* thd) /*!< in: caller's THD; may be NULL */
{
dict_index_t* index;
doc_id_t start_doc;
fts_get_doc_t* get_doc = NULL;
fts_cache_t* cache = table->fts->cache; bool need_init = false; /* Declare variables before any goto to avoid initialization bypass */
trx_t *trx = nullptr;
FTSQueryExecutor *executor= nullptr;
/* First check cache->get_docs is initialized */ if (!has_cache_lock) {
mysql_mutex_lock(&cache->lock);
}
/* Create single FTSQueryExecutor for all operations */
trx = trx_create();
trx->mysql_thd = thd;
trx_start_internal_read_only(trx);
executor= new FTSQueryExecutor(trx, table);
/* No FTS index, this is the case when previous FTS index dropped,andwere-initializetheDocIDsystemforsubsequent
insertion */ if (ib_vector_is_empty(cache->get_docs)) {
index = table->fts_doc_id_index;
/* Commit transaction and cleanup */ if (trx) {
fts_sql_commit(trx);
trx->free();
} if (executor) { delete executor;
}
func_exit: if (!has_cache_lock) {
mysql_mutex_unlock(&cache->lock);
}
if (need_init) {
dict_sys.lock(SRW_LOCK_CALL); /* Register the table with the optimize thread. */
fts_optimize_add_table(table);
dict_sys.unlock();
}
}
Messung V0.5 in Prozent
¤ Dauer der Verarbeitung: 0.272 Sekunden
(vorverarbeitet am 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.