YoushouldhavereceivedacopyoftheGNUGeneralPublicLicense alongwiththisprogram;ifnot,writetotheFreeSoftware
Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1335 USA */
/** @brief type of checkpoint currently running */ static CHECKPOINT_LEVEL checkpoint_in_progress= CHECKPOINT_NONE; /** @brief protects checkpoint_in_progress */ static mysql_mutex_t LOCK_checkpoint; /** @brief for killing the background checkpoint thread */ static mysql_cond_t COND_checkpoint; /** @brief control structure for checkpoint background thread */ static MA_SERVICE_THREAD_CONTROL checkpoint_control=
{0, FALSE, FALSE, &LOCK_checkpoint, &COND_checkpoint}; /* is ulong like pagecache->blocks_changed */ static uint pages_to_flush_before_next_checkpoint; static PAGECACHE_FILE *dfiles, /**< data files to flush in background */
*dfiles_end; /**< list of data files ends here */ static PAGECACHE_FILE *kfiles, /**< index files to flush in background */
*kfiles_end; /**< list of index files ends here */
struct st_filter_param
{
LSN up_to_lsn; /**< only pages with rec_lsn < this LSN */
uint max_pages; /**< stop after flushing this number pages */
}; /**< information to determine which dirty pages should be flushed */
err:
error= 1;
my_printf_error(HA_ERR_GENERIC, "Aria engine: checkpoint failed at %s with " "error %d", MYF(ME_ERROR_LOG),
error_place, (error_errno ? error_errno : my_errno)); /* we were possibly not able to determine what pages to flush */
pages_to_flush_before_next_checkpoint= 0;
end: if (record_pieces)
{ for (i= 0; i < record_pieces_count; i++)
my_free(record_pieces[i].str);
}
my_afree(record_pieces);
mysql_mutex_lock(&LOCK_checkpoint);
checkpoint_in_progress= CHECKPOINT_NONE;
mysql_mutex_unlock(&LOCK_checkpoint);
DBUG_RETURN(error);
}
static ulong maria_checkpoint_min_cache_activity= 10*1024*1024; /* Set in ha_maria.cc */
uint maria_checkpoint_min_log_activity= 1*1024*1024;
pthread_handler_t ma_checkpoint_background(void *arg)
{ /** @brief At least this of log/page bytes written between checkpoints */ /* Iftheintervalcouldbechangedbytheuserwhileweareinthisthread, itcouldbeannoying:forexampleitcouldcause"case2"tobeexecuted rightafter"case0",thushaving'dfile'unset.Sothethreadcaresonly abouttheinterval'svaluewhenitstarted.
*/ const size_t interval= (size_t)arg;
size_t sleeps, sleep_time;
TRANSLOG_ADDRESS log_horizon_at_last_checkpoint=
translog_get_horizon();
ulonglong pagecache_flushes_at_last_checkpoint=
multi_global_cache_writes(&maria_pagecaches);
uint UNINIT_VAR(pages_bunch_size); struct st_filter_param filter_param;
PAGECACHE_FILE *UNINIT_VAR(dfile); /**< data file currently being flushed */
PAGECACHE_FILE *UNINIT_VAR(kfile); /**< index file currently being flushed */
for(;;) /* iterations of checkpoints and dirty page flushing */
{ #if0/* good for testing, to do a lot of checkpoints, finds a lot of bugs */
sleeps=0; #endif switch (sleeps % interval)
{ case0:
{ /* If checkpoints are disabled, wait 1 second and try again */ if (maria_checkpoint_disabled)
{
sleep_time= 1; break;
}
{
TRANSLOG_ADDRESS horizon= translog_get_horizon();
ulonglong writes= multi_global_cache_writes(&maria_pagecaches);
/* Withbackgroundflushingevenlydistributedoverthetime betweentwocheckpoints,weshouldhaveonlylittleflushingtodo inthecheckpoint.
*/ /* Nocheckpointiflittleworkofinterestforrecoverywasdone sincelastcheckpoint.Suchworkincludeslogwriting(lengthens recovery,checkpointwouldshortenit),pageflushing(checkpoint woulddecreasetheamountofreadpagesinrecovery). Incaseofoneshortstatementperminute(verylowload),wedon't wanttocheckpointeveryminute,hencethepositive maria_checkpoint_min_activity.
*/ if ((ulonglong) (horizon - log_horizon_at_last_checkpoint) <=
maria_checkpoint_min_log_activity &&
((ulonglong) (writes - pagecache_flushes_at_last_checkpoint) *
maria_pagecaches.caches->block_size) <=
maria_checkpoint_min_cache_activity)
{ /* Notenoughhashappendsincelastcheckpoint. Sleepforawhileandtryagainlater
*/
sleep_time= interval; break;
}
sleep_time= 1;
ma_checkpoint_execute(CHECKPOINT_MEDIUM, TRUE); /* Snapshotthiskindof"state"oftheengine.Notethatthevalue belowispossiblygreaterthanlast_checkpoint_lsn.
*/
log_horizon_at_last_checkpoint= translog_get_horizon();
pagecache_flushes_at_last_checkpoint= writes; /* Ifthecheckpointabovesucceededithassetd|kfilesand d|kfiles_end.Ifishasfailed,ithasset pages_to_flush_before_next_checkpointto0sowewillskipflushing andsleepuntilthenextcheckpoint.
*/
} break;
} case1: /* set up parameters for background page flushing */
filter_param.up_to_lsn= last_checkpoint_lsn;
pages_bunch_size= pages_to_flush_before_next_checkpoint / (uint)interval;
dfile= dfiles;
kfile= kfiles; /* fall through */ default: if (pages_bunch_size > 0)
{
DBUG_PRINT("checkpoint",
("Maria background checkpoint thread: %u pages",
pages_bunch_size)); /* flush a bunch of dirty pages */
filter_param.max_pages= pages_bunch_size; while (dfile != dfiles_end)
{ /* WeuseFLUSH_KEEP_LAZY:ifafileisalreadyinflush,it's smartertomovetothenextfilethanwaitforthisonetobe completelyflushed,whichmaytakelong. StaleFilePointersInFlush:noticehowbelowweuse"dfile"which isanOSfiledescriptorplussomefunctionandMARIA_SHARE pointers;thisdatadatesfromapreviouscheckpoint;sincethen, thetablemayhavebeenclosed(soMARIA_SHARE*becamestale),and thefiledescriptorreassignedtoanothertablewhichdoesnot havethesameCRC-read-setcallbacks:itisthusimportantthat flush_pagecache_blocks_with_filter()doesnotusethepointers, onlytheOSfiledescriptor.
*/ int res=
flush_pagecache_blocks_with_filter(dfile->pagecache,
dfile, FLUSH_KEEP_LAZY,
filter_flush_file_evenly,
&filter_param); if (unlikely(res & PCFLUSH_ERROR))
ma_message_no_user(0, "background data page flush failed"); if (filter_param.max_pages == 0) /* bunch all flushed, sleep */ break; /* and we will continue with the same file */
dfile++; /* otherwise all this file is flushed, move to next file */ /* MikaelRnotedthatheobservedthatLinux'sfilecachemaynever fsynctodiskuntilthiscacheisfull,atwhichpointitdecides toemptythecache,makingthemachineveryslow.Asolutionwas tofsyncafterwriting2MB.Sowemightwanttofsync()hereif wewroteenoughpages.
*/
} while (kfile != kfiles_end)
{ int res=
flush_pagecache_blocks_with_filter(kfile->pagecache,
kfile, FLUSH_KEEP_LAZY,
filter_flush_file_evenly,
&filter_param); if (unlikely(res & PCFLUSH_ERROR))
ma_message_no_user(0, "background index page flush failed"); if (filter_param.max_pages == 0) /* bunch all flushed, sleep */ break; /* and we will continue with the same file */
kfile++; /* otherwise all this file is flushed, move to next file */
}
sleep_time= 1;
} else
{ /* Can directly sleep until the next checkpoint moment */
sleep_time= interval - (sleeps % interval);
}
} if (my_service_thread_sleep(&checkpoint_control,
sleep_time * 1000000000ULL)) break;
sleeps+= sleep_time;
}
DBUG_PRINT("info",("Maria background checkpoint thread ends"));
{
CHECKPOINT_LEVEL level= CHECKPOINT_FULL; /* That'sthefinalone,whichguaranteesthatacleanshutdownalwaysends withacheckpoint.
*/
DBUG_EXECUTE_IF("maria_checkpoint_indirect", level= CHECKPOINT_INDIRECT;);
ma_checkpoint_execute(level, FALSE);
}
my_thread_end(); return0;
}
str->length= 4 + /* number of tables */
(2 + /* short id */
LSN_STORE_SIZE + /* first_log_write_at_lsn */ 1/* end-of-name 0 */
) * nb + total_names_length; if (unlikely((str->str= my_malloc(PSI_INSTRUMENT_ME, str->length, MYF(MY_WME))) == NULL)) goto err;
ptr= str->str;
ptr+= 4; /* real number of stored tables is not yet know */
/* only possible checkpointer, so can do the read below without mutex */
filter_param.up_to_lsn= last_checkpoint_lsn; switch(checkpoint_in_progress)
{ case CHECKPOINT_MEDIUM:
filter= &filter_flush_file_medium; break; case CHECKPOINT_FULL:
filter= &filter_flush_file_full; break; case CHECKPOINT_INDIRECT:
filter= NULL; break; default:
DBUG_ASSERT(0); goto err;
}
for (nb_stored= 0, i= 0; i < nb; i++)
{
MARIA_SHARE *share= distinct_shares[i];
PAGECACHE_FILE kfile, dfile;
my_bool ignore_share; if (!(share->in_checkpoint & MARIA_CHECKPOINT_LOOKS_AT_ME))
{ /* Noneedforamutextoreadtheabove,onlyuscanwrite*this*bitof thein_checkpointbitmap
*/ continue;
} /** @todoWeshouldnotlookattableswhichdidn'tchangesincelast checkpoint.
*/
DBUG_PRINT("info",("looking at table '%s'", share->open_file_name.str)); if (state_copy == state_copies_end) /* we have no more cached states */
{ /* Collectandcacheabunchofstates.Wedothisformanystatesata time,tonotlock/unlockthelog'slocktoooften.
*/
uint j, bound= MY_MIN(nb, i + STATE_COPIES);
state_copy= state_copies; /* part of the state is protected by log's lock */
translog_lock();
state_copies_horizon= translog_get_horizon_no_lock(); for (j= i; j < bound; j++)
{
MARIA_SHARE *share2= distinct_shares[j]; if (!(share2->in_checkpoint & MARIA_CHECKPOINT_LOOKS_AT_ME)) continue;
state_copy->index= j;
state_copy->state= share2->state; /* we copy the state */
state_copy++; /* data_file_lengthisnotupdatedunderlog'slockbythebitmap code,butwritingawrongdata_file_lengthisok:anext maria_close()willcorrectit;ifwecrashbefore,Recoverywill setittothetruephysicalsize.
*/
}
translog_unlock(); if (state_copy == state_copies) break; /* Nothing to do */
/** Wearegoingtoflushthesestates. Before,allrecordsdescribinghowtoundosuchstatemustbe inthelog(WAL).UsuallythismeansUNDOs.Inthespecialcaseof data|key_file_length,recoveryjustneedstoopenthetabletofixthe length,soanyLOGREC_FILE_ID/REDO/UNDOallowingrecoveryto understanditmustopenatable,isenough;soaslongas data|key_file_lengthisupdatedafterwritinganylogrecordit'sok: ifwecopiednewvalueabove,itmeanstherecordwasbefore state_copies_horizonandweflushsuchrecordbelow. Apartfromdata|key_file_lengthwhichareeasilyrecoverablefromthe realfile'ssize,allotherstatemembersmustbeupdatedonlywhen writingtheUNDO;otherwise,ifupdatedbefore,iftheirnewvalueis flushedbyacheckpointandthereisacrashbeforeUNDOiswritten, theirREDOgroupwillbemissingoratleastincompleteandskipped byrecovery,sobadstatevaluewillstay.Forexample,setting key_rootbeforewritingtheUNDO:thetablewouldhaveoldindex pages(theywerepinnedattimeofcrash)andanew,thuswrong, key_root. @todoRECOVERYBUGcheckthatallcodehonoursthat.
*/ if (translog_flush(state_copies_horizon)) goto err; /* now we have cached states and they are WAL-safe*/
state_copies_end= state_copy-1;
state_copy= state_copies;
}
/* locate our state among these cached ones */ for ( ; state_copy->index != i; state_copy++)
DBUG_ASSERT(state_copy <= state_copies_end);
/* OS file descriptors are ints which we stored in 4 bytes */
compile_time_assert(sizeof(int) <= 4); /* Protectagainstmaria_close()(whichdoessomememoryfreeingin MARIA_FILE_BITMAP)withclose_lock.intern_lockisnot sufficientaswe,aswellasmaria_close(),aregoingtounlock intern_lockinthemiddleofmanipulatingthetable.Serializingusand maria_close()shouldhelpavoidproblems.
*/
mysql_mutex_lock(&share->close_lock);
mysql_mutex_lock(&share->intern_lock); /* Tablesinanormalstatehavetheirtwofiledescriptorsopen. InsomerarecaseslikeREPAIR,somedescriptormaybeclosedoreven -1.Ifthathappened,the_ma_state_info_write()mayfail.Thisis preventedbyenclosingallplaceswhichclose/changekfile.filewith intern_lock.
*/
kfile= share->kfile;
dfile= share->bitmap.file; /* Ignoretablewhichhasnologgedwrites(allitsfuturelogrecordswill befoundnaturallybyRecovery).Ignoreobsoleteshares(_before_ settingthemselvestolast_version=0theyalreadydidallflushand sync;ifweflushtheirstatenowwemaybeflushinganobsoletestate ontoanewerone(assumingthetablehasbeenreopenedwithadifferent sharebutofcoursesamephysicalindexfile).
*/
ignore_share= (share->id == 0) | (share->last_version == 0);
DBUG_PRINT("info", ("ignore_share: %d", ignore_share)); if (!ignore_share)
{
size_t open_file_name_len= share->open_file_name.length + 1; /* remember the descriptors for background flush */
*(dfiles_end++)= dfile;
*(kfiles_end++)= kfile; /* we will store this table in the record */
nb_stored++;
int2store(ptr, share->id);
ptr+= 2;
lsn_store(ptr, share->lsn_of_file_id);
ptr+= LSN_STORE_SIZE; /* first_bitmap_with_spaceisnotupdatedunderlog'slock,andis important.Wewouldneedthebitmap'slocktogetitright.Recovery ofthisisnotclear,sowejustplaysafe:writeitoutas unknown:ifcrash,_ma_bitmap_init()atnextopen(forexamplein Recovery)willconvertitto0andthusthefirstinsertionwill searchforfreespacefromthefile'sfirstbitmap(0)- under-optimalbutsafe. Ifnocrash,maria_close()willwritetheexactvalue.
*/
state_copy->state.first_bitmap_with_space= ~(ulonglong)0;
memcpy(ptr, share->open_file_name.str, open_file_name_len);
ptr+= open_file_name_len; if (cmp_translog_addr(share->state.is_of_horizon,
checkpoint_start_log_horizon) >= 0)
{ /* Statewasflushedrecently,itdoesnotholddownthelog's low-watermarkandwillnotgiveavoidableworktoRecovery.Sowe needn'tflushit.Also,itispossiblethatwhilewecopiedthe stateabove(underlog'slock,withoutintern_lock)itwasbeing modifiedinmemoryorflushedtodisk(withoutlog'slock,under intern_lock,likeinmaria_extra()),soourcopymaybeincorrect andweshouldnotflushit. Itmayalsobeasharewhichgotlast_version==0sincewechecked last_version;inthiscase,itflusheditsstateandtheLSNtest abovewillcatchit.
*/
} else
{ /* Wecoulddothestateflushonlyifshare->changed,butit's tricky. Consideramaria_write()whichhaswrittenREDO,UNDO,andbeforeit calls_ma_writeinfo()(settingshare->changed=1),checkpoint happensandseesshare->changed=0,doesnotflushstate.Itis possiblethatRecoverydoesnotstartfrombeforetheREDOandthus thestateisnotrecovered.Asolutionmaybetoset share->changed=1underlogmutexwhenwritinglogrecords.
if (!ignore_share)
{ if (filter != NULL)
{ if ((flush_pagecache_blocks_with_filter(dfile.pagecache,
&dfile, FLUSH_KEEP_LAZY,
filter, &filter_param) &
PCFLUSH_ERROR))
ma_message_no_user(0, "checkpoint data page flush failed"); if ((flush_pagecache_blocks_with_filter(kfile.pagecache,
&kfile, FLUSH_KEEP_LAZY,
filter, &filter_param) &
PCFLUSH_ERROR))
ma_message_no_user(0, "checkpoint index page flush failed");
} /* fsyncsthefd,that'stheloooongoperation(e.g.max150fsync persecond,soifyouhavetouched1000filesit's7seconds).
*/
sync_error|=
mysql_file_sync(dfile.file, MYF(MY_WME | MY_IGNORE_BADFD)) |
mysql_file_sync(kfile.file, MYF(MY_WME | MY_IGNORE_BADFD)); /* incaseoferror,wecontinuebecausewritingothertablestodiskis stilluseful.
*/
}
}
if (sync_error) goto err; /* We maybe over-estimated (due to share->id==0 or last_version==0) */
DBUG_ASSERT(str->length >= (uint)(ptr - str->str));
str->length= (uint)(ptr - str->str); /* Aswesupportmax65ktablesopenatatime(2-byteshortid),we assumeuintisenoughforthecumulatedlengthoftablenames;and LEX_STRING::lengthisuint.
*/
int4store(str->str, nb_stored);
error= unmark_tables= 0;
err: if (unlikely(unmark_tables))
{ /* maria_close() uses THR_LOCK_maria from start to end */
mysql_mutex_lock(&THR_LOCK_maria); for (i= 0; i < nb; i++)
{
MARIA_SHARE *share= distinct_shares[i]; if (share->in_checkpoint & MARIA_CHECKPOINT_SHOULD_FREE_ME)
{
share->in_checkpoint&= ~MARIA_CHECKPOINT_SHOULD_FREE_ME; /* maria_close() left us to free the share */
free_maria_share(share);
} else
{ /* share goes back to normal state */
share->in_checkpoint= 0;
}
}
mysql_mutex_unlock(&THR_LOCK_maria);
}
my_free(distinct_shares);
my_free(state_copies);
DBUG_RETURN(error);
}
Messung V0.5 in Prozent
¤ Dauer der Verarbeitung: 0.31 Sekunden
(vorverarbeitet am 2026-10-08)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.