diff --git a/docs/memory_management.md b/docs/memory_management.md index 41290a613..921431136 100644 --- a/docs/memory_management.md +++ b/docs/memory_management.md @@ -23,12 +23,11 @@ The dominant in-memory cost of a live dense index is the HNSW structure: - the base layer, allocated as `maxElements * sizeDataAtBaseLayer_` - upper-layer node storage in `dataUpperLayer_` -- the vector cache, sized from `VECTOR_CACHE_PERCENTAGE` and `VECTOR_CACHE_MIN_BITS` - the visited-list pool and other small bookkeeping structures One important detail: Endee does not load the full dense vector corpus into the HNSW object. -Dense vectors stay in `VectorStorage` and are fetched on demand through the vector fetcher and the -vector cache. So the main DRAM cost is the graph plus cache, not a second full copy of the vector +Dense vectors stay in `VectorStorage` and are fetched on demand from disk through the vector +fetcher. So the main DRAM cost is the graph itself, not a second full copy of the vector database. ## Scaling diff --git a/src/hnsw/hnswalg.h b/src/hnsw/hnswalg.h index 7cde9be8b..b90200816 100644 --- a/src/hnsw/hnswalg.h +++ b/src/hnsw/hnswalg.h @@ -22,7 +22,6 @@ #include "visited_list_pool.h" #include "hnswlib.h" -#include "vector_cache.h" #include "log.hpp" #include "../utils/settings.hpp" #include "../quant/dispatch.hpp" @@ -116,15 +115,6 @@ namespace hnswlib { << data_size_ << ", dimension: " << dimension_ << ", quant_level: " << static_cast(quant_level_)); - // Initialize cache - size_t cache_bits = VectorCache::calculateCacheBits(maxElements_); - if (cache_bits > 0) { - vector_cache_ = std::make_unique(data_size_, cache_bits); - LOG_DEBUG("Vector cache initialized for " << maxElements_ << " elements with " << (1ULL << cache_bits) << " slots"); - size_t cache_bytes = vector_cache_->getMemoryUsage(); - LOG_DEBUG("Vector cache allocated: " << cache_bytes << " bytes (" << (cache_bytes / MB) << " MB)"); - } - LOG_DEBUG("Unified layer data size: " << data_size_); // M_ cannot be more than settings::MAX_M @@ -206,16 +196,9 @@ namespace hnswlib { size += upper_layer_estimate * (data_size_ + sizeof(levelInt) + sizeLinksUpperLayers_); - if (vector_cache_) { - size += vector_cache_->getMemoryUsage(); - } - return size / GB; // GB } - // Cache management getters/setters - // Removed as cache is managed externally - template std::vector> searchKnn(const void* query_data, @@ -450,14 +433,6 @@ namespace hnswlib { fstSimFunc_ = space_->get_sim_func(); dist_func_param_ = space_->get_dist_func_param(); - // Initialize cache for loaded index - size_t cache_bits = VectorCache::calculateCacheBits(maxElements_); - LOG_INFO(2101, "Calculated cache bits for loaded index: " << cache_bits); - if (cache_bits > 0) { - vector_cache_ = std::make_unique(data_size_, cache_bits); - LOG_DEBUG("Vector cache initialized for " << maxElements_ << " elements with " << (1ULL << cache_bits) << " slots"); - } - // Allocate memory and load level 0 data dataBaseLayer_ = (char*)malloc(maxElements_ * sizeDataAtBaseLayer_); if(dataBaseLayer_ == nullptr) { @@ -533,9 +508,6 @@ namespace hnswlib { if(visited_list_pool_ == nullptr) { throw std::runtime_error("Not enough memory"); } - - // Adjust cache based on element count and cache percentage threshold (default - // VECTOR_CACHE_PERCENTAGE) adjustCacheForElementCount(curElementsCount_); } @@ -611,16 +583,6 @@ namespace hnswlib { return; } } - // Update cached value only if this id is already present in cache. - // Do not insert on add/update path; cold ids can be populated on read hot path. - if constexpr(!is_new) { - if(vector_cache_) { - // Fast atomic invalidation instead of taking a cache lock that can deadlock. - // The cache will automatically refresh this vector on the next read miss. - vector_cache_->invalidateSlot(cur_c); - } - } - // std::unique_lock lock_el(getLinkListMutex(cur_c)); // Initialize level 0 links @@ -831,15 +793,9 @@ namespace hnswlib { SIMFUNC fstSimFunc_; void* dist_func_param_{nullptr}; - // Cache for vectors - mutable std::unique_ptr vector_cache_; - static constexpr size_t CACHE_LOCK_STRIPE_BITS = 10; // 1024 striped locks in HNSW - static constexpr size_t CACHE_LOCK_STRIPE_COUNT = 1 << CACHE_LOCK_STRIPE_BITS; - static constexpr size_t CACHE_LOCK_STRIPE_MASK = CACHE_LOCK_STRIPE_COUNT - 1; - mutable std::array vectorCacheLocks_; - - struct CacheReadView { - std::shared_lock lock; + // Zero-copy view over an in-memory vector (upper-layer blob). Vectors that + // only live on disk are read into a caller-provided buffer instead. + struct ZeroCopyView { const uint8_t* data = nullptr; explicit operator bool() const { @@ -847,19 +803,7 @@ namespace hnswlib { } }; - std::shared_mutex& getCacheMutex(idhInt internal_id) const { - size_t cache_index = static_cast(internal_id); - if(vector_cache_) { - cache_index = vector_cache_->getCacheIndex(internal_id); - } - size_t stripe_id = cache_index & CACHE_LOCK_STRIPE_MASK; - return vectorCacheLocks_[stripe_id]; - } - public: - const VectorCache* getCache() const { - return vector_cache_.get(); - } // Maps external label to internal id std::vector labelLookup_; @@ -901,63 +845,31 @@ namespace hnswlib { return dataUpperLayer_[internal_id].get(); } - // Modified function returning bool and filling buffer. - // For upper-layer vectors, data is returned zero-copy via cache_read_handle only. - // For layer 0, callers can optionally receive zero-copy cache hits via cache_read_handle. + // Returns true and fills the result. For in-memory upper-layer vectors the data + // is returned zero-copy via data_view (when provided). For layer 0 the vector is + // read directly from disk into the caller-provided buffer. bool getDataByInternalId(idhInt internal_id, levelInt layer, uint8_t* buffer, - CacheReadView* cache_read_handle = nullptr) const { - if(cache_read_handle) { - *cache_read_handle = CacheReadView(); + ZeroCopyView* data_view = nullptr) const { + if(data_view) { + *data_view = ZeroCopyView(); } const uint8_t* upper_ptr = getUpperLayerDataPtr(internal_id); if(upper_ptr) { - if(cache_read_handle) { - cache_read_handle->data = upper_ptr; + if(data_view) { + data_view->data = upper_ptr; return true; } return false; } if(layer == 0) { - // Check cache first - if(vector_cache_) { - if(cache_read_handle) { - cache_read_handle->lock = - std::shared_lock(getCacheMutex(internal_id)); - const uint8_t* cached_ptr = vector_cache_->getPointer(internal_id); - if(cached_ptr) { - cache_read_handle->data = cached_ptr; - return true; - } - cache_read_handle->lock.unlock(); - } - - if(buffer) { - std::shared_lock lock(getCacheMutex(internal_id)); - const uint8_t* cached_ptr = vector_cache_->getPointer(internal_id); - if(cached_ptr) { - memcpy(buffer, cached_ptr, data_size_); - return true; - } - } - } - + // Not held in memory; read straight from disk into the caller buffer. idInt external_label = getExternalLabel(internal_id); if(vector_fetcher_ && buffer) { - // Directly fetch to buffer - bool success = vector_fetcher_(external_label, buffer); - - // Populate cache on successful fetch - if (success && vector_cache_) { - std::unique_lock write_lock(getCacheMutex(internal_id), std::try_to_lock); - if(write_lock.owns_lock()) { - vector_cache_->insert(internal_id, buffer); - } - } - return success; + return vector_fetcher_(external_label, buffer); } return false; } @@ -966,7 +878,8 @@ namespace hnswlib { return false; } - // Batch fetch for level 0: check upper-layer data/cache first, then fetch misses in one MDBX txn. + // Batch fetch for level 0: resolve in-memory upper-layer data first, then fetch + // the remaining disk misses in one MDBX txn. // internal_ids: array of internal IDs to fetch // buffers: flat output buffer, count * data_size_ bytes // success: output array of bools @@ -975,7 +888,7 @@ namespace hnswlib { void getDataByInternalIdBatch(const idhInt* internal_ids, uint8_t* buffers, bool* success, size_t count, const void** data_ptrs = nullptr) const { - // Phase 1: Check cache for all IDs, collect misses + // Phase 1: resolve in-memory upper-layer hits, collect disk misses std::vector miss_indices; // index into the batch std::vector miss_labels; // external labels for MDBX lookup miss_indices.reserve(count); @@ -998,19 +911,6 @@ namespace hnswlib { continue; } - if(vector_cache_) { - std::shared_lock lock(getCacheMutex(internal_ids[i])); - const uint8_t* cached_ptr = vector_cache_->getPointer(internal_ids[i]); - if(cached_ptr) { - std::memcpy(buf, cached_ptr, data_size_); - if(data_ptrs) { - data_ptrs[i] = buf; - } - success[i] = true; - continue; - } - } - success[i] = false; miss_indices.push_back(i); miss_labels.push_back(getExternalLabel(internal_ids[i])); @@ -1026,7 +926,7 @@ namespace hnswlib { vector_fetcher_batch_(miss_labels.data(), miss_buffers.data(), miss_success.get(), miss_indices.size()); - // Phase 3: Copy results back and populate cache + // Phase 3: Copy results back for(size_t mi = 0; mi < miss_indices.size(); mi++) { size_t i = miss_indices[mi]; if(miss_success[mi]) { @@ -1036,14 +936,6 @@ namespace hnswlib { data_ptrs[i] = buf; } success[i] = true; - // Populate cache - if(vector_cache_) { - std::unique_lock write_lock( - getCacheMutex(internal_ids[i]), std::try_to_lock); - if(write_lock.owns_lock()) { - vector_cache_->insert(internal_ids[i], buf); - } - } } } } else if(!miss_indices.empty() && vector_fetcher_) { @@ -1056,13 +948,6 @@ namespace hnswlib { if(ok && data_ptrs) { data_ptrs[i] = buf; } - if(ok && vector_cache_) { - std::unique_lock write_lock( - getCacheMutex(internal_ids[i]), std::try_to_lock); - if(write_lock.owns_lock()) { - vector_cache_->insert(internal_ids[i], buf); - } - } } } } @@ -1204,14 +1089,14 @@ namespace hnswlib { } const void* cand_vec = nullptr; - CacheReadView cand_cache_handle; + ZeroCopyView cand_view; if(level == 0) { if(getDataByInternalId(candidate.second, level, cand_buf.data(), - &cand_cache_handle)) { - cand_vec = cand_cache_handle - ? static_cast(cand_cache_handle.data) + &cand_view)) { + cand_vec = cand_view + ? static_cast(cand_view.data) : static_cast(cand_buf.data()); } } else { @@ -1336,14 +1221,14 @@ namespace hnswlib { setListCount(ll_other, sz + 1); } else { const void* neighbor_data = nullptr; - CacheReadView neighbor_cache_handle; + ZeroCopyView neighbor_view; if(level == 0) { if(getDataByInternalId(neighbor, level, neighbor_buf.data(), - &neighbor_cache_handle)) { - neighbor_data = neighbor_cache_handle - ? static_cast(neighbor_cache_handle.data) + &neighbor_view)) { + neighbor_data = neighbor_view + ? static_cast(neighbor_view.data) : static_cast(neighbor_buf.data()); } } else { @@ -1363,14 +1248,14 @@ namespace hnswlib { for(size_t j = 0; j < sz; j++) { dist_t sim; const void* other_neighbor_data = nullptr; - CacheReadView other_cache_handle; + ZeroCopyView other_view; if(level == 0) { if(getDataByInternalId(data[j], level, data_buf.data(), - &other_cache_handle)) { - other_neighbor_data = other_cache_handle - ? static_cast(other_cache_handle.data) + &other_view)) { + other_neighbor_data = other_view + ? static_cast(other_view.data) : static_cast(data_buf.data()); } } else { @@ -1441,14 +1326,14 @@ namespace hnswlib { dist_t sim = std::numeric_limits::lowest(); if(!has_deletions || !isMarkedDeleted(ep_id)) { const void* vec_data = nullptr; - CacheReadView ep_cache_handle; + ZeroCopyView ep_view; if(layer == 0) { if(getDataByInternalId(ep_id, layer, buffer.data(), - &ep_cache_handle)) { - vec_data = ep_cache_handle - ? static_cast(ep_cache_handle.data) + &ep_view)) { + vec_data = ep_view + ? static_cast(ep_view.data) : static_cast(buffer.data()); } } else { diff --git a/src/hnsw/vector_cache.h b/src/hnsw/vector_cache.h deleted file mode 100644 index e5188b602..000000000 --- a/src/hnsw/vector_cache.h +++ /dev/null @@ -1,241 +0,0 @@ -// Endee — high-performance vector database -// Copyright (C) 2026 Endee Labs -// -// Portions of this file are derived from hnswlib (https://github.com/nmslib/hnswlib), -// originally licensed under the Apache License 2.0. As incorporated and modified -// here, the combined work is distributed under the GNU Affero General Public License. -// -// This program is free software: you can redistribute it and/or modify -// it under the terms of the GNU Affero General Public License as published by -// the Free Software Foundation, either version 3 of the License, or -// (at your option) any later version. -// -// This program is distributed in the hope that it will be useful, -// but WITHOUT ANY WARRANTY; without even the implied warranty of -// MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the -// GNU Affero General Public License for more details. -// -// You should have received a copy of the GNU Affero General Public License -// along with this program. If not, see . - -#pragma once -#include "hnswlib.h" -#include "../utils/settings.hpp" -#include -#include -#include -#include -#include -#include -#include - -namespace hnswlib { - -class VectorCache { -public: - - // Helper to calculate required cache bits based on element count and percentage - static size_t calculateCacheBits(size_t element_count, size_t cache_percent = settings::VECTOR_CACHE_PERCENTAGE) { - if (element_count == 0 || cache_percent == 0) return 0; - - size_t target_elements = (element_count * cache_percent) / 100; - - // Calculate bits needed: 2^bits >= target_elements - size_t cache_bits = 0; - while ((1ULL << cache_bits) < target_elements) { - cache_bits++; - } - - // Enforce minimum bits - if (cache_bits < settings::VECTOR_CACHE_MIN_BITS) { - cache_bits = settings::VECTOR_CACHE_MIN_BITS; - } - - return cache_bits; - } - -private: - size_t cacheBits_ = 0; - size_t cacheSize_ = 0; - size_t cacheMask_ = 0; - size_t vectorCacheDataSize_ = 0; - size_t data_size_ = 0; - uint8_t* vectorCache_ = nullptr; - std::atomic* slotLife_ = nullptr; - - static constexpr idInt INVALID_ID = static_cast(-1); - - static constexpr uint8_t SLOT_LIFE_INVALID = 0; - static constexpr uint8_t SLOT_LIFE_COLD = 1; - static constexpr uint8_t SLOT_LIFE_WARM = 2; - static constexpr uint8_t SLOT_LIFE_HOT = 3; - -public: - VectorCache() = default; - - // Constructor with initialization - VectorCache(size_t data_size, size_t cache_bits) { - init(data_size, cache_bits); - } - - ~VectorCache() { - if (vectorCache_) { - delete[] vectorCache_; - vectorCache_ = nullptr; - } - if (slotLife_) { - delete[] slotLife_; - slotLife_ = nullptr; - } - } - - void init(size_t data_size, size_t cache_bits) { - if (vectorCache_) { - delete[] vectorCache_; - vectorCache_ = nullptr; - } - if (slotLife_) { - delete[] slotLife_; - slotLife_ = nullptr; - } - - if (cache_bits == 0) { - cacheBits_ = 0; - cacheSize_ = 0; - cacheMask_ = 0; - data_size_ = 0; - vectorCacheDataSize_ = 0; - return; - } - - data_size_ = data_size; - cacheBits_ = cache_bits; - cacheSize_ = 1 << cacheBits_; - cacheMask_ = cacheSize_ - 1; - vectorCacheDataSize_ = data_size_ + sizeof(idInt); - - vectorCache_ = new uint8_t[cacheSize_ * vectorCacheDataSize_]; - slotLife_ = new std::atomic[cacheSize_]; - - // Initialize all entries to INVALID_ID - for (size_t i = 0; i < cacheSize_; i++) { - idInt* id_ptr = reinterpret_cast(vectorCache_ + i * vectorCacheDataSize_); - *id_ptr = INVALID_ID; - slotLife_[i].store(SLOT_LIFE_INVALID, std::memory_order_relaxed); - } - } - - size_t getCacheIndex(idInt internal_id) const { - return internal_id & cacheMask_; - } - - // Not thread-safe. Caller must hold the appropriate cache stripe lock. - const uint8_t* getPointer(idInt internal_id) { - if (!vectorCache_ || !slotLife_) return nullptr; - - size_t index = getCacheIndex(internal_id); - - // If life is invalid, we treat it as a miss regardless of ID - if (slotLife_[index].load(std::memory_order_relaxed) == SLOT_LIFE_INVALID) { - return nullptr; - } - - uint8_t* entry = vectorCache_ + index * vectorCacheDataSize_; - idInt* stored_id = reinterpret_cast(entry); - if (*stored_id != internal_id) { - return nullptr; - } - - slotLife_[index].store(SLOT_LIFE_HOT, std::memory_order_relaxed); - return entry + sizeof(idInt); - } - - // Not thread-safe. Caller must hold the appropriate cache stripe lock. - const uint8_t* getPointer(idInt internal_id) const { - return const_cast(this)->getPointer(internal_id); - } - - // Not thread-safe. Caller must hold the appropriate cache stripe lock. - void insert(idInt internal_id, const uint8_t* data) { - if (!vectorCache_ || !slotLife_) return; - - size_t index = getCacheIndex(internal_id); - uint8_t* entry = vectorCache_ + index * vectorCacheDataSize_; - - idInt* stored_id = reinterpret_cast(entry); - - // Same id: refresh value and restore life. - if (*stored_id == internal_id) { - memcpy(entry + sizeof(idInt), data, data_size_); - slotLife_[index].store(SLOT_LIFE_HOT, std::memory_order_relaxed); - return; - } - - // Different id: three-life policy (eviction protocol). - // life == SLOT_LIFE_HOT (3) -> demote to WARM (2), do not replace now. - // life == SLOT_LIFE_WARM (2) -> demote to COLD (1), do not replace now. - // life == SLOT_LIFE_COLD (1) or INVALID (0) -> replace slot and set life to WARM (2) - // Newly inserted items start as WARM. They must be accessed again to become HOT. - uint8_t life = slotLife_[index].load(std::memory_order_relaxed); - if (life == SLOT_LIFE_HOT) { - slotLife_[index].store(SLOT_LIFE_WARM, std::memory_order_relaxed); - return; - } else if (life == SLOT_LIFE_WARM) { - slotLife_[index].store(SLOT_LIFE_COLD, std::memory_order_relaxed); - return; - } - - *stored_id = internal_id; - memcpy(entry + sizeof(idInt), data, data_size_); - slotLife_[index].store(SLOT_LIFE_WARM, std::memory_order_relaxed); - } - - // Not thread-safe. Caller must hold the appropriate cache stripe lock. - void update(idInt internal_id, const uint8_t* data) { - if(!vectorCache_ || !slotLife_) return; - - size_t index = getCacheIndex(internal_id); - uint8_t* entry = vectorCache_ + index * vectorCacheDataSize_; - idInt* stored_id = reinterpret_cast(entry); - - if(*stored_id != internal_id) { - return; - } - - *stored_id = internal_id; - memcpy(entry + sizeof(idInt), data, data_size_); - slotLife_[index].store(SLOT_LIFE_COLD, std::memory_order_relaxed); - } - - // Atomically invalidate the slot for a given id, forcing a fetch on the next read. - // Thread-safe without a cache stripe lock due to atomic memory ordering, - // provided the caller accepts eventual consistency (next reader will miss and lock to fetch). - void invalidateSlot(idInt internal_id) { - if(!vectorCache_ || !slotLife_) return; - - size_t index = getCacheIndex(internal_id); - - // Only invalidate if the slot currently belongs to this ID, preventing us from accidentally - // invalidating another vector if an eviction happened between our read and invalidate. - uint8_t* entry = vectorCache_ + index * vectorCacheDataSize_; - idInt* stored_id = reinterpret_cast(entry); - - if (*stored_id == internal_id) { - // Note: Data race risk with readers. Even if we set slotLife to INVALID right now, - // a reader might have just checked slotLife and is now about to read `stored_id`. - // Because of this, we set slotLife so FUTURE readers instantly miss. - slotLife_[index].store(SLOT_LIFE_INVALID, std::memory_order_release); - } - } - - size_t getCacheBits() const { return cacheBits_; } - size_t getCacheSize() const { return cacheSize_; } - void setCacheBits(size_t bits) { cacheBits_ = bits; } - - size_t getMemoryUsage() const { - if (!vectorCache_) return 0; - return cacheSize_ * vectorCacheDataSize_; - } -}; - -} // namespace hnswlib diff --git a/src/utils/settings.hpp b/src/utils/settings.hpp index 78455fb10..3772fddf7 100644 --- a/src/utils/settings.hpp +++ b/src/utils/settings.hpp @@ -128,8 +128,6 @@ namespace settings { constexpr size_t DEFAULT_MAX_ELEMENTS = 100'000; constexpr size_t DEFAULT_MAX_ELEMENTS_INCREMENT = 100'000; constexpr size_t DEFAULT_MAX_ELEMENTS_INCREMENT_TRIGGER = 50'000; - constexpr size_t DEFAULT_VECTOR_CACHE_PERCENTAGE = 50; - constexpr size_t DEFAULT_VECTOR_CACHE_MIN_BITS = 17; // Minimum 128K entries in cache const std::string DEFAULT_SERVER_ID = "unknown"; //For Backups @@ -193,16 +191,6 @@ namespace settings { return env ? std::stoull(env) : DEFAULT_MAX_ELEMENTS_INCREMENT_TRIGGER; }(); - inline static size_t VECTOR_CACHE_PERCENTAGE = [] { - const char* env = std::getenv("NDD_VECTOR_CACHE_PERCENTAGE"); - return env ? std::min(std::stoull(env), 100) : DEFAULT_VECTOR_CACHE_PERCENTAGE; - }(); - - inline static size_t VECTOR_CACHE_MIN_BITS = [] { - const char* env = std::getenv("NDD_VECTOR_CACHE_MIN_BITS"); - return env ? std::stoull(env) : DEFAULT_VECTOR_CACHE_MIN_BITS; - }(); - inline static size_t MAX_LIVE_INDICES = [] { const char* env = std::getenv("NDD_MAX_LIVE_INDICES"); return env ? std::stoull(env) : DEFAULT_MAX_LIVE_INDICES;