Continuum C++ API
Unified runtime for token + tensor execution
Loading...
Searching...
No Matches
continuum::runtime::KVCacheIndex Class Reference

#include <kv_prefix_cache.hpp>

Classes

struct  SnapshotEntry
 
struct  TrieNode
 

Public Member Functions

 KVCacheIndex (std::size_t max_entries=8192)
 
std::optional< std::pair< CacheEntry, std::int32_t > > longest_prefix (const std::string &model_id, const DecodeParams &decode, const std::vector< std::int32_t > &tokens, const std::string &cache_namespace={}) const
 
void insert (CacheEntry entry, const std::vector< std::int32_t > &token_prefix)
 
void insert_unlocked (CacheEntry entry, const std::vector< std::int32_t > &token_prefix)
 
void invalidate (void *backend_handle)
 
void clear ()
 
std::size_t size () const
 
std::size_t max_entries () const
 Capacity in entries passed at construction.
 
std::size_t estimated_bytes () const
 
bool save_metadata (const std::string &path) const
 
bool load_metadata (const std::string &path)
 
std::vector< SnapshotEntry > snapshot () const
 

Detailed Description

Prefix-KV tier: a token trie of reusable backend states.

Eviction: least-recently-used. Every trie depth covered by an insert holds its own entry, so size() counts per-depth entries. insert and a longest_prefix hit refresh recency; once size() exceeds max_entries the least recently used entry is removed and empty branches are compacted.

Constructor & Destructor Documentation

◆ KVCacheIndex()

continuum::runtime::KVCacheIndex::KVCacheIndex ( std::size_t  max_entries = 8192)
explicit

Member Function Documentation

◆ clear()

void continuum::runtime::KVCacheIndex::clear ( )

◆ estimated_bytes()

std::size_t continuum::runtime::KVCacheIndex::estimated_bytes ( ) const

Approximate resident bytes of the index itself: trie nodes plus entry metadata. Backend-owned state behind each handle is not counted.

◆ insert()

void continuum::runtime::KVCacheIndex::insert ( CacheEntry  entry,
const std::vector< std::int32_t > &  token_prefix 
)

◆ insert_unlocked()

void continuum::runtime::KVCacheIndex::insert_unlocked ( CacheEntry  entry,
const std::vector< std::int32_t > &  token_prefix 
)

◆ invalidate()

void continuum::runtime::KVCacheIndex::invalidate ( void *  backend_handle)

◆ load_metadata()

bool continuum::runtime::KVCacheIndex::load_metadata ( const std::string &  path)

◆ longest_prefix()

std::optional< std::pair< CacheEntry, std::int32_t > > continuum::runtime::KVCacheIndex::longest_prefix ( const std::string &  model_id,
const DecodeParams &  decode,
const std::vector< std::int32_t > &  tokens,
const std::string &  cache_namespace = {} 
) const

◆ max_entries()

std::size_t continuum::runtime::KVCacheIndex::max_entries ( ) const
inline

Capacity in entries passed at construction.

◆ save_metadata()

bool continuum::runtime::KVCacheIndex::save_metadata ( const std::string &  path) const

◆ size()

std::size_t continuum::runtime::KVCacheIndex::size ( ) const

◆ snapshot()

std::vector< SnapshotEntry > continuum::runtime::KVCacheIndex::snapshot ( ) const

The documentation for this class was generated from the following file: