|
Continuum C++ API
Unified runtime for token + tensor execution
|
#include <semantic_cache.hpp>
Classes | |
| struct | LookupResult |
Public Member Functions | |
| SemanticCacheIndex (std::size_t max_entries=2048, float similarity_threshold=0.85f) | |
| LookupResult | lookup (const std::vector< float > &query_embedding, const std::string &model_id, const std::string &cache_namespace={}, const std::string &embedder_id={}, const std::string &query_prompt={}) const |
| Best entry for the same model, namespace, and embedder identity. | |
| void | insert (const std::vector< float > &embedding, const std::string &model_id, std::vector< std::uint8_t > output, const std::string &cache_namespace={}, const std::string &embedder_id={}, const std::string &prompt={}) |
| void | set_verifier (std::shared_ptr< const HitVerifier > verifier) |
| Replace the hit verifier; nullptr serves any candidate above threshold. | |
| std::shared_ptr< const HitVerifier > | verifier () const |
| std::int64_t | verifier_rejections () const |
| Candidates turned down by the verifier since construction / clear(). | |
| void | clear () |
| std::size_t | size () const |
| std::size_t | max_entries () const |
| Capacity in entries passed at construction. | |
| std::size_t | estimated_bytes () const |
| Approximate resident bytes: embeddings, outputs, ids, and per-entry overhead. | |
| float | similarity_threshold () const |
| void | set_similarity_threshold (float t) |
Static Public Member Functions | |
| static float | cosine_similarity (const std::vector< float > &a, const std::vector< float > &b) |
Paraphrase-tolerant tier keyed by embedding similarity.
Eviction: least-recently-used. insert stamps an entry and a lookup that clears the threshold refreshes the matched entry; when size() reaches max_entries the least recently used entry is dropped before inserting.
Verification: candidates that clear the threshold are checked, best first, by a HitVerifier (default: LexicalNearMissVerifier) against the prompt they were cached for; the first one it accepts is served. The check needs both prompt texts, so it applies when insert and lookup are given them (the interpreter always does). set_verifier(nullptr) disables it.
|
explicit |
| void continuum::runtime::SemanticCacheIndex::clear | ( | ) |
|
static |
| std::size_t continuum::runtime::SemanticCacheIndex::estimated_bytes | ( | ) | const |
Approximate resident bytes: embeddings, outputs, ids, and per-entry overhead.
| void continuum::runtime::SemanticCacheIndex::insert | ( | const std::vector< float > & | embedding, |
| const std::string & | model_id, | ||
| std::vector< std::uint8_t > | output, | ||
| const std::string & | cache_namespace = {}, |
||
| const std::string & | embedder_id = {}, |
||
| const std::string & | prompt = {} |
||
| ) |
| LookupResult continuum::runtime::SemanticCacheIndex::lookup | ( | const std::vector< float > & | query_embedding, |
| const std::string & | model_id, | ||
| const std::string & | cache_namespace = {}, |
||
| const std::string & | embedder_id = {}, |
||
| const std::string & | query_prompt = {} |
||
| ) | const |
Best entry for the same model, namespace, and embedder identity.
|
inline |
Capacity in entries passed at construction.
| void continuum::runtime::SemanticCacheIndex::set_similarity_threshold | ( | float | t | ) |
| void continuum::runtime::SemanticCacheIndex::set_verifier | ( | std::shared_ptr< const HitVerifier > | verifier | ) |
Replace the hit verifier; nullptr serves any candidate above threshold.
| float continuum::runtime::SemanticCacheIndex::similarity_threshold | ( | ) | const |
| std::size_t continuum::runtime::SemanticCacheIndex::size | ( | ) | const |
| std::shared_ptr< const HitVerifier > continuum::runtime::SemanticCacheIndex::verifier | ( | ) | const |
| std::int64_t continuum::runtime::SemanticCacheIndex::verifier_rejections | ( | ) | const |
Candidates turned down by the verifier since construction / clear().