// helper function for combining the status of two memory contexts // useful for implementing hybrid memory types (e.g. iSWA)
llama_memory_status llama_memory_status_combine(llama_memory_status s0, llama_memory_status s1);
// helper function for checking if a memory status indicates a failure bool llama_memory_status_is_fail(llama_memory_status status);
// the interface for managing the memory context during batch processing // this interface is implemented per memory type. see: // - llama_kv_cache_unified_context // - llama_kv_cache_unified_iswa_context // ... // // the only method that should mutate the memory and the memory context is llama_memory_i::apply() struct llama_memory_context_i { virtual ~llama_memory_context_i() = default;
// consume the current ubatch from the context and proceed to the next one // return false if we are done virtualbool next() = 0;
// apply the memory state for the current ubatch to the memory object // return false on failure virtualbool apply() = 0;
// get the current ubatch virtualconst llama_ubatch & get_ubatch() const = 0;
// get the status of the memory context - used for error handling and checking if any updates would be applied virtual llama_memory_status get_status() const = 0;
};
using llama_memory_context_ptr = std::unique_ptr<llama_memory_context_i>;
// general concept of LLM memory // the KV cache is a type of LLM memory, but there can be other types struct llama_memory_i { virtual ~llama_memory_i() = default;
// split the input batch into a set of ubatches and verify that they can fit into the cache // return a context object containing the ubatches and memory state required to process them // check the llama_memory_context_i::get_status() for the result virtual llama_memory_context_ptr init_batch(
llama_batch_allocr & balloc,
uint32_t n_ubatch, bool embd_all) = 0;
// simulate full cache, used for allocating worst-case compute buffers virtual llama_memory_context_ptr init_full() = 0;
// prepare for any pending memory updates, such as shifts, defrags, etc. // status == LLAMA_MEMORY_STATUS_NO_UPDATE if there is nothing to update virtual llama_memory_context_ptr init_update(llama_context * lctx, bool optimize) = 0;
// getters virtualbool get_can_shift() const = 0;
// // ops //
// if data == true, the data buffers will also be cleared together with the metadata virtualvoid clear(bool data) = 0;
using llama_memory_ptr = std::unique_ptr<llama_memory_i>;
// TODO: temporary until the llama_kv_cache is removed from the public API struct llama_kv_cache : public llama_memory_i { virtual ~llama_kv_cache() = default;
};
Messung V0.5 in Prozent
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.0.17Bemerkung:
(vorverarbeitet am 2026-09-28)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.