#include <cassert>
#include <cinttypes>
#include <cstddef>
+#include <cstdint>
#include <exception>
#include <functional>
#include <numeric>
(void)old_meta;
}
-inline bool ClockUpdate(ClockHandle& h, bool* purgeable = nullptr) {
+inline bool ClockUpdate(ClockHandle& h, BaseClockTable::EvictionData* data,
+ bool* purgeable = nullptr) {
uint64_t meta;
if (purgeable) {
assert(*purgeable == false);
(meta >> ClockHandle::kReleaseCounterShift) & ClockHandle::kCounterMask;
if (acquire_count != release_count) {
// Only clock update entries with no outstanding refs
+ data->seen_pinned_count++;
return false;
}
if ((meta >> ClockHandle::kStateShift == ClockHandle::kStateVisible) &&
<< ClockHandle::kStateShift) |
(meta & ClockHandle::kHitBitMask))) {
// Took ownership.
+ data->freed_charge += h.GetTotalCharge();
+ data->freed_count += 1;
return true;
} else {
// Compare-exchange failing probably
}
}
+constexpr uint32_t kStrictCapacityLimitBit = 1u << 31;
+
+uint32_t SanitizeEncodeEecAndScl(int eviction_effort_cap,
+ bool strict_capacit_limit) {
+ eviction_effort_cap = std::max(int{1}, eviction_effort_cap);
+ eviction_effort_cap =
+ std::min(static_cast<int>(~kStrictCapacityLimitBit), eviction_effort_cap);
+ uint32_t eec_and_scl = static_cast<uint32_t>(eviction_effort_cap);
+ eec_and_scl |= strict_capacit_limit ? kStrictCapacityLimitBit : 0;
+ return eec_and_scl;
+}
+
} // namespace
void ClockHandleBasicData::FreeData(MemoryAllocator* allocator) const {
template <class Table>
typename Table::HandleImpl* BaseClockTable::CreateStandalone(
- ClockHandleBasicData& proto, size_t capacity, bool strict_capacity_limit,
+ ClockHandleBasicData& proto, size_t capacity, uint32_t eec_and_scl,
bool allow_uncharged) {
Table& derived = static_cast<Table&>(*this);
typename Table::InsertState state;
derived.StartInsert(state);
const size_t total_charge = proto.GetTotalCharge();
- if (strict_capacity_limit) {
+ // NOTE: we can use eec_and_scl as eviction_effort_cap below because
+ // strict_capacity_limit=true is supposed to disable the limit on eviction
+ // effort, and a large value effectively does that.
+ if (eec_and_scl & kStrictCapacityLimitBit) {
Status s = ChargeUsageMaybeEvictStrict<Table>(
total_charge, capacity,
- /*need_evict_for_occupancy=*/false, state);
+ /*need_evict_for_occupancy=*/false, eec_and_scl, state);
if (!s.ok()) {
if (allow_uncharged) {
proto.total_charge = 0;
// Case strict_capacity_limit == false
bool success = ChargeUsageMaybeEvictNonStrict<Table>(
total_charge, capacity,
- /*need_evict_for_occupancy=*/false, state);
+ /*need_evict_for_occupancy=*/false, eec_and_scl, state);
if (!success) {
// Force the issue
usage_.FetchAddRelaxed(total_charge);
template <class Table>
Status BaseClockTable::ChargeUsageMaybeEvictStrict(
size_t total_charge, size_t capacity, bool need_evict_for_occupancy,
- typename Table::InsertState& state) {
+ uint32_t eviction_effort_cap, typename Table::InsertState& state) {
if (total_charge > capacity) {
return Status::MemoryLimit(
"Cache entry too large for a single cache shard: " +
}
if (request_evict_charge > 0) {
EvictionData data;
- static_cast<Table*>(this)->Evict(request_evict_charge, state, &data);
+ static_cast<Table*>(this)->Evict(request_evict_charge, state, &data,
+ eviction_effort_cap);
occupancy_.FetchSub(data.freed_count);
if (LIKELY(data.freed_charge > need_evict_charge)) {
assert(data.freed_count > 0);
template <class Table>
inline bool BaseClockTable::ChargeUsageMaybeEvictNonStrict(
size_t total_charge, size_t capacity, bool need_evict_for_occupancy,
- typename Table::InsertState& state) {
+ uint32_t eviction_effort_cap, typename Table::InsertState& state) {
// For simplicity, we consider that either the cache can accept the insert
// with no evictions, or we must evict enough to make (at least) enough
// space. It could lead to unnecessary failures or excessive evictions in
}
EvictionData data;
if (need_evict_charge > 0) {
- static_cast<Table*>(this)->Evict(need_evict_charge, state, &data);
+ static_cast<Table*>(this)->Evict(need_evict_charge, state, &data,
+ eviction_effort_cap);
// Deal with potential occupancy deficit
if (UNLIKELY(need_evict_for_occupancy) && data.freed_count == 0) {
assert(data.freed_charge == 0);
return true;
}
-void BaseClockTable::TrackAndReleaseEvictedEntry(
- ClockHandle* h, BaseClockTable::EvictionData* data) {
- data->freed_charge += h->GetTotalCharge();
- data->freed_count += 1;
-
+void BaseClockTable::TrackAndReleaseEvictedEntry(ClockHandle* h) {
bool took_value_ownership = false;
if (eviction_callback_) {
// For key reconstructed from hash
MarkEmpty(*h);
}
+bool IsEvictionEffortExceeded(const BaseClockTable::EvictionData& data,
+ uint32_t eviction_effort_cap) {
+ // Basically checks whether the ratio of useful effort to wasted effort is
+ // too low, with a start-up allowance for wasted effort before any useful
+ // effort.
+ return (data.freed_count + 1U) * uint64_t{eviction_effort_cap} <=
+ data.seen_pinned_count;
+}
+
template <class Table>
Status BaseClockTable::Insert(const ClockHandleBasicData& proto,
typename Table::HandleImpl** handle,
Cache::Priority priority, size_t capacity,
- bool strict_capacity_limit) {
+ uint32_t eec_and_scl) {
using HandleImpl = typename Table::HandleImpl;
Table& derived = static_cast<Table&>(*this);
// strict_capacity_limit, but mostly pessimistic.
bool use_standalone_insert = false;
const size_t total_charge = proto.GetTotalCharge();
- if (strict_capacity_limit) {
+ // NOTE: we can use eec_and_scl as eviction_effort_cap below because
+ // strict_capacity_limit=true is supposed to disable the limit on eviction
+ // effort, and a large value effectively does that.
+ if (eec_and_scl & kStrictCapacityLimitBit) {
Status s = ChargeUsageMaybeEvictStrict<Table>(
- total_charge, capacity, need_evict_for_occupancy, state);
+ total_charge, capacity, need_evict_for_occupancy, eec_and_scl, state);
if (!s.ok()) {
// Revert occupancy
occupancy_.FetchSubRelaxed(1);
} else {
// Case strict_capacity_limit == false
bool success = ChargeUsageMaybeEvictNonStrict<Table>(
- total_charge, capacity, need_evict_for_occupancy, state);
+ total_charge, capacity, need_evict_for_occupancy, eec_and_scl, state);
if (!success) {
// Revert occupancy
occupancy_.FetchSubRelaxed(1);
#endif
FixedHyperClockTable::FixedHyperClockTable(
- size_t capacity, bool /*strict_capacity_limit*/,
- CacheMetadataChargePolicy metadata_charge_policy,
+ size_t capacity, CacheMetadataChargePolicy metadata_charge_policy,
MemoryAllocator* allocator,
const Cache::EvictionCallback* eviction_callback, const uint32_t* hash_seed,
const Opts& opts)
}
inline void FixedHyperClockTable::Evict(size_t requested_charge, InsertState&,
- EvictionData* data) {
+ EvictionData* data,
+ uint32_t eviction_effort_cap) {
// precondition
assert(requested_charge > 0);
for (;;) {
for (size_t i = 0; i < step_size; i++) {
HandleImpl& h = array_[ModTableSize(Lower32of64(old_clock_pointer + i))];
- bool evicting = ClockUpdate(h);
+ bool evicting = ClockUpdate(h, data);
if (evicting) {
Rollback(h.hashed_key, &h);
- TrackAndReleaseEvictedEntry(&h, data);
+ TrackAndReleaseEvictedEntry(&h);
}
}
if (old_clock_pointer >= max_clock_pointer) {
return;
}
+ if (IsEvictionEffortExceeded(*data, eviction_effort_cap)) {
+ eviction_effort_exceeded_count_.FetchAddRelaxed(1);
+ return;
+ }
// Advance clock pointer (concurrently)
old_clock_pointer = clock_pointer_.FetchAddRelaxed(step_size);
const Cache::EvictionCallback* eviction_callback, const uint32_t* hash_seed,
const typename Table::Opts& opts)
: CacheShardBase(metadata_charge_policy),
- table_(capacity, strict_capacity_limit, metadata_charge_policy, allocator,
- eviction_callback, hash_seed, opts),
+ table_(capacity, metadata_charge_policy, allocator, eviction_callback,
+ hash_seed, opts),
capacity_(capacity),
- strict_capacity_limit_(strict_capacity_limit) {
+ eec_and_scl_(SanitizeEncodeEecAndScl(opts.eviction_effort_cap,
+ strict_capacity_limit)) {
// Initial charge metadata should not exceed capacity
assert(table_.GetUsage() <= capacity_.LoadRelaxed() ||
capacity_.LoadRelaxed() < sizeof(HandleImpl));
template <class Table>
void ClockCacheShard<Table>::SetStrictCapacityLimit(
bool strict_capacity_limit) {
- strict_capacity_limit_.StoreRelaxed(strict_capacity_limit);
+ if (strict_capacity_limit) {
+ eec_and_scl_.FetchOrRelaxed(kStrictCapacityLimitBit);
+ } else {
+ eec_and_scl_.FetchAndRelaxed(~kStrictCapacityLimitBit);
+ }
// next Insert will take care of any necessary evictions
}
proto.total_charge = charge;
return table_.template Insert<Table>(proto, handle, priority,
capacity_.LoadRelaxed(),
- strict_capacity_limit_.LoadRelaxed());
+ eec_and_scl_.LoadRelaxed());
}
template <class Table>
proto.value = obj;
proto.helper = helper;
proto.total_charge = charge;
- return table_.template CreateStandalone<Table>(
- proto, capacity_.LoadRelaxed(), strict_capacity_limit_.LoadRelaxed(),
- allow_uncharged);
+ return table_.template CreateStandalone<Table>(proto, capacity_.LoadRelaxed(),
+ eec_and_scl_.LoadRelaxed(),
+ allow_uncharged);
}
template <class Table>
const std::shared_ptr<Logger>& info_log) const {
if (info_log->GetInfoLogLevel() <= InfoLogLevel::DEBUG_LEVEL) {
LoadVarianceStats slot_stats;
+ uint64_t eviction_effort_exceeded_count = 0;
this->ForEachShard([&](const BaseHyperClockCache<Table>::Shard* shard) {
size_t count = shard->GetTableAddressCount();
for (size_t i = 0; i < count; ++i) {
slot_stats.Add(IsSlotOccupied(*shard->GetTable().HandlePtr(i)));
}
+ eviction_effort_exceeded_count +=
+ shard->GetTable().GetEvictionEffortExceededCount();
});
ROCKS_LOG_AT_LEVEL(info_log, InfoLogLevel::DEBUG_LEVEL,
"Slot occupancy stats: %s", slot_stats.Report().c_str());
+ ROCKS_LOG_AT_LEVEL(info_log, InfoLogLevel::DEBUG_LEVEL,
+ "Eviction effort exceeded: %" PRIu64,
+ eviction_effort_exceeded_count);
}
}
};
AutoHyperClockTable::AutoHyperClockTable(
- size_t capacity, bool /*strict_capacity_limit*/,
- CacheMetadataChargePolicy metadata_charge_policy,
+ size_t capacity, CacheMetadataChargePolicy metadata_charge_policy,
MemoryAllocator* allocator,
const Cache::EvictionCallback* eviction_callback, const uint32_t* hash_seed,
const Opts& opts)
template <class OpData>
void AutoHyperClockTable::PurgeImplLocked(OpData* op_data,
ChainRewriteLock& rewrite_lock,
- size_t home) {
+ size_t home,
+ BaseClockTable::EvictionData* data) {
constexpr bool kIsPurge = std::is_same_v<OpData, PurgeLockedOpData>;
constexpr bool kIsClockUpdateChain =
std::is_same_v<OpData, ClockUpdateChainLockedOpData>;
assert(home == BottomNBits(h->hashed_key[1], home_shift));
if constexpr (kIsClockUpdateChain) {
// Clock update and/or check for purgeable (under (de)construction)
- if (ClockUpdate(*h, &purgeable)) {
+ if (ClockUpdate(*h, data, &purgeable)) {
// Remember for finishing eviction
op_data->push_back(h);
// Entries for eviction become purgeable
}
} else {
(void)op_data;
+ (void)data;
purgeable = ((h->meta.Load() >> ClockHandle::kStateShift) &
ClockHandle::kStateShareableBit) == 0;
}
using ClockUpdateChainOpData = ClockUpdateChainLockedOpData;
template <class OpData>
-void AutoHyperClockTable::PurgeImpl(OpData* op_data, size_t home) {
+void AutoHyperClockTable::PurgeImpl(OpData* op_data, size_t home,
+ BaseClockTable::EvictionData* data) {
// Early efforts to make AutoHCC fully wait-free ran into too many problems
// that needed obscure and potentially inefficient work-arounds to have a
// chance at working.
if (!rewrite_lock.IsEnd()) {
if constexpr (kIsPurge) {
PurgeLockedOpData* locked_op_data{};
- PurgeImplLocked(locked_op_data, rewrite_lock, home);
+ PurgeImplLocked(locked_op_data, rewrite_lock, home, data);
} else {
- PurgeImplLocked(op_data, rewrite_lock, home);
+ PurgeImplLocked(op_data, rewrite_lock, home, data);
}
}
}
}
void AutoHyperClockTable::Evict(size_t requested_charge, InsertState& state,
- EvictionData* data) {
+ EvictionData* data,
+ uint32_t eviction_effort_cap) {
// precondition
assert(requested_charge > 0);
if (home >= used_length) {
break;
}
- PurgeImpl(&to_finish_eviction, home);
+ PurgeImpl(&to_finish_eviction, home, data);
}
}
for (HandleImpl* h : to_finish_eviction) {
- TrackAndReleaseEvictedEntry(h, data);
+ TrackAndReleaseEvictedEntry(h);
// NOTE: setting likely_empty_slot here can cause us to reduce the
// portion of "at home" entries, probably because an evicted entry
// is more likely to come back than a random new entry and would be
if (old_clock_pointer + step_size >= max_clock_pointer) {
return;
}
+
+ if (IsEvictionEffortExceeded(*data, eviction_effort_cap)) {
+ eviction_effort_exceeded_count_.FetchAddRelaxed(1);
+ return;
+ }
}
}
#include <array>
#include <atomic>
+#include <climits>
#include <cstddef>
#include <cstdint>
#include <memory>
class BaseClockTable {
public:
+ struct BaseOpts {
+ explicit BaseOpts(int _eviction_effort_cap)
+ : eviction_effort_cap(_eviction_effort_cap) {}
+ explicit BaseOpts(const HyperClockCacheOptions& opts)
+ : BaseOpts(opts.eviction_effort_cap) {}
+ int eviction_effort_cap;
+ };
+
BaseClockTable(CacheMetadataChargePolicy metadata_charge_policy,
MemoryAllocator* allocator,
const Cache::EvictionCallback* eviction_callback,
template <class Table>
typename Table::HandleImpl* CreateStandalone(ClockHandleBasicData& proto,
size_t capacity,
- bool strict_capacity_limit,
+ uint32_t eec_and_scl,
bool allow_uncharged);
template <class Table>
Status Insert(const ClockHandleBasicData& proto,
typename Table::HandleImpl** handle, Cache::Priority priority,
- size_t capacity, bool strict_capacity_limit);
+ size_t capacity, uint32_t eec_and_scl);
void Ref(ClockHandle& handle);
uint64_t GetYieldCount() const { return yield_count_.LoadRelaxed(); }
+ uint64_t GetEvictionEffortExceededCount() const {
+ return eviction_effort_exceeded_count_.LoadRelaxed();
+ }
+
struct EvictionData {
size_t freed_charge = 0;
size_t freed_count = 0;
+ size_t seen_pinned_count = 0;
};
- void TrackAndReleaseEvictedEntry(ClockHandle* h, EvictionData* data);
+ void TrackAndReleaseEvictedEntry(ClockHandle* h);
#ifndef NDEBUG
// Acquire N references
template <class Table>
Status ChargeUsageMaybeEvictStrict(size_t total_charge, size_t capacity,
bool need_evict_for_occupancy,
+ uint32_t eviction_effort_cap,
typename Table::InsertState& state);
// Helper for updating `usage_` for new entry with given `total_charge`
template <class Table>
bool ChargeUsageMaybeEvictNonStrict(size_t total_charge, size_t capacity,
bool need_evict_for_occupancy,
+ uint32_t eviction_effort_cap,
typename Table::InsertState& state);
protected: // data
RelaxedAtomic<uint64_t> clock_pointer_{};
// Counter for number of times we yield to wait on another thread.
+ // It is normal for this to occur rarely in normal operation.
// (Relaxed: a simple stat counter.)
RelaxedAtomic<uint64_t> yield_count_{};
+ // Counter for number of times eviction effort cap is exceeded.
+ // It is normal for this to occur rarely in normal operation.
+ // (Relaxed: a simple stat counter.)
+ RelaxedAtomic<uint64_t> eviction_effort_exceeded_count_{};
+
// TODO: is this separation needed if we don't do background evictions?
ALIGN_AS(CACHE_LINE_SIZE)
// Number of elements in the table.
inline void SetStandalone() { standalone = true; }
}; // struct HandleImpl
- struct Opts {
- explicit Opts(size_t _estimated_value_size)
- : estimated_value_size(_estimated_value_size) {}
- explicit Opts(const HyperClockCacheOptions& opts) {
+ struct Opts : public BaseOpts {
+ explicit Opts(size_t _estimated_value_size, int _eviction_effort_cap)
+ : BaseOpts(_eviction_effort_cap),
+ estimated_value_size(_estimated_value_size) {}
+ explicit Opts(const HyperClockCacheOptions& opts)
+ : BaseOpts(opts.eviction_effort_cap) {
assert(opts.estimated_entry_charge > 0);
estimated_value_size = opts.estimated_entry_charge;
}
size_t estimated_value_size;
};
- FixedHyperClockTable(size_t capacity, bool strict_capacity_limit,
+ FixedHyperClockTable(size_t capacity,
CacheMetadataChargePolicy metadata_charge_policy,
MemoryAllocator* allocator,
const Cache::EvictionCallback* eviction_callback,
// Runs the clock eviction algorithm trying to reclaim at least
// requested_charge. Returns how much is evicted, which could be less
// if it appears impossible to evict the requested amount without blocking.
- void Evict(size_t requested_charge, InsertState& state, EvictionData* data);
+ void Evict(size_t requested_charge, InsertState& state, EvictionData* data,
+ uint32_t eviction_effort_cap);
HandleImpl* Lookup(const UniqueId64x2& hashed_key);
}
}; // struct HandleImpl
- struct Opts {
- explicit Opts(size_t _min_avg_value_size)
- : min_avg_value_size(_min_avg_value_size) {}
+ struct Opts : public BaseOpts {
+ explicit Opts(size_t _min_avg_value_size, int _eviction_effort_cap)
+ : BaseOpts(_eviction_effort_cap),
+ min_avg_value_size(_min_avg_value_size) {}
- explicit Opts(const HyperClockCacheOptions& opts) {
+ explicit Opts(const HyperClockCacheOptions& opts)
+ : BaseOpts(opts.eviction_effort_cap) {
assert(opts.estimated_entry_charge == 0);
min_avg_value_size = opts.min_avg_entry_charge;
}
size_t min_avg_value_size;
};
- AutoHyperClockTable(size_t capacity, bool strict_capacity_limit,
+ AutoHyperClockTable(size_t capacity,
CacheMetadataChargePolicy metadata_charge_policy,
MemoryAllocator* allocator,
const Cache::EvictionCallback* eviction_callback,
// Runs the clock eviction algorithm trying to reclaim at least
// requested_charge. Returns how much is evicted, which could be less
// if it appears impossible to evict the requested amount without blocking.
- void Evict(size_t requested_charge, InsertState& state, EvictionData* data);
+ void Evict(size_t requested_charge, InsertState& state, EvictionData* data,
+ uint32_t eviction_effort_cap);
HandleImpl* Lookup(const UniqueId64x2& hashed_key);
// with proper handling to ensure all existing data is seen even in the
// presence of concurrent insertions, etc. (See implementation.)
template <class OpData>
- void PurgeImpl(OpData* op_data, size_t home = SIZE_MAX);
+ void PurgeImpl(OpData* op_data, size_t home = SIZE_MAX,
+ EvictionData* data = nullptr);
// An RAII wrapper for locking a chain of entries for removals. See
// implementation.
// implementation.
template <class OpData>
void PurgeImplLocked(OpData* op_data, ChainRewriteLock& rewrite_lock,
- size_t home);
+ size_t home, EvictionData* data);
// Update length_info_ as much as possible without waiting, given a known
// usable (ready for inserts and lookups) grow_home. (Previous grow_homes
// (Relaxed: eventual consistency/update is OK)
RelaxedAtomic<size_t> capacity_;
- // Whether to reject insertion if cache reaches its full capacity.
+ // Encodes eviction_effort_cap (bottom 31 bits) and strict_capacity_limit
+ // (top bit). See HyperClockCacheOptions::eviction_effort_cap etc.
// (Relaxed: eventual consistency/update is OK)
- RelaxedAtomic<bool> strict_capacity_limit_;
+ RelaxedAtomic<uint32_t> eec_and_scl_;
}; // class ClockCacheShard
template <class Table>
}
}
- void NewShard(size_t capacity, bool strict_capacity_limit = true) {
+ void NewShard(size_t capacity, bool strict_capacity_limit = true,
+ int eviction_effort_cap = 30) {
DeleteShard();
shard_ =
reinterpret_cast<Shard*>(port::cacheline_aligned_alloc(sizeof(Shard)));
- TableOpts opts{1 /*value_size*/};
+ TableOpts opts{1 /*value_size*/, eviction_effort_cap};
new (shard_)
Shard(capacity, strict_capacity_limit, kDontChargeCacheMetadata,
/*allocator*/ nullptr, &eviction_callback_, &hash_seed_, opts);
return Slice(reinterpret_cast<const char*>(&hashed_key), 16U);
}
+ // A bad hash function for testing / stressing collision handling
static inline UniqueId64x2 TestHashedKey(char key) {
// For testing hash near-collision behavior, put the variance in
// hashed_key in bits that are unlikely to be used as hash bits.
return {(static_cast<uint64_t>(key) << 56) + 1234U, 5678U};
}
+ // A reasonable hash function, for testing "typical behavior" etc.
+ template <typename T>
+ static inline UniqueId64x2 CheapHash(T i) {
+ return {static_cast<uint64_t>(i) * uint64_t{0x85EBCA77C2B2AE63},
+ static_cast<uint64_t>(i) * uint64_t{0xC2B2AE3D27D4EB4F}};
+ }
+
Shard* shard_ = nullptr;
private:
}
}
+TYPED_TEST(ClockCacheTest, ClockEvictionEffortCapTest) {
+ using HandleImpl = typename ClockCacheTest<TypeParam>::Shard::HandleImpl;
+ for (bool strict_capacity_limit : {true, false}) {
+ SCOPED_TRACE("strict_capacity_limit = " +
+ std::to_string(strict_capacity_limit));
+ for (int eec : {-42, 0, 1, 10, 100, 1000}) {
+ SCOPED_TRACE("eviction_effort_cap = " + std::to_string(eec));
+ constexpr size_t kCapacity = 1000;
+ // Start with much larger capacity to ensure that we can go way over
+ // capacity without reaching table occupancy limit.
+ this->NewShard(3 * kCapacity, strict_capacity_limit, eec);
+ auto& shard = *this->shard_;
+ shard.SetCapacity(kCapacity);
+
+ // Nearly fill the cache with pinned entries, then add a bunch of
+ // non-pinned entries. eviction_effort_cap should affect how many
+ // evictable entries are present beyond the cache capacity, despite
+ // being evictable.
+ constexpr size_t kCount = kCapacity - 1;
+ std::unique_ptr<HandleImpl* []> ha { new HandleImpl* [kCount] {} };
+ for (size_t i = 0; i < 2 * kCount; ++i) {
+ UniqueId64x2 hkey = this->CheapHash(i);
+ ASSERT_OK(shard.Insert(
+ this->TestKey(hkey), hkey, nullptr /*value*/, &kNoopCacheItemHelper,
+ 1 /*charge*/, i < kCount ? &ha[i] : nullptr, Cache::Priority::LOW));
+ }
+
+ if (strict_capacity_limit) {
+ // If strict_capacity_limit is enabled, the cache will never exceed its
+ // capacity
+ EXPECT_EQ(shard.GetOccupancyCount(), kCapacity);
+ } else {
+ // Rough inverse relationship between cap and possible memory
+ // explosion, which shows up as increased table occupancy count.
+ int effective_eec = std::max(int{1}, eec) + 1;
+ EXPECT_NEAR(shard.GetOccupancyCount() * 1.0,
+ kCount * (1 + 1.4 / effective_eec),
+ kCount * (0.6 / effective_eec) + 1.0);
+ }
+
+ for (size_t i = 0; i < kCount; ++i) {
+ shard.Release(ha[i]);
+ }
+ }
+ }
+}
+
namespace {
struct DeleteCounter {
int deleted = 0;
// to find the appropriate balance automatically.
// * Cache priorities are less aggressively enforced, which could cause
// cache dilution from long range scans (unless they use fill_cache=false).
-// * Can be worse for small caches, because if almost all of a cache shard is
-// pinned (more likely with non-partitioned filters), then CLOCK eviction
-// becomes very CPU intensive.
//
// See internal cache/clock_cache.h for full description.
struct HyperClockCacheOptions : public ShardedCacheOptions {
// load factor for efficient Lookup, Insert, etc.
size_t min_avg_entry_charge = 450;
+ // A tuning parameter to cap eviction CPU usage in a "thrashing" situation
+ // by allowing the memory capacity to be exceeded slightly as needed. The
+ // default setting should offer balanced protection against excessive CPU
+ // and memory usage under extreme stress conditions, with no effect on
+ // normal operation. Such stress conditions are proportionally more likely
+ // with small caches (10s of MB or less) vs. large caches (GB-scale).
+ // (NOTE: With the unusual setting of strict_capacity_limit=true, this
+ // parameter is ignored.)
+ //
+ // BACKGROUND: Without some kind of limiter, inserting into a CLOCK-based
+ // cache with no evictable entries (all "pinned") requires scanning the
+ // entire cache to determine that nothing can be evicted. (By contrast,
+ // LRU caches can determine no entries are evictable in O(1) time, but
+ // require more synchronization/coordination on that eviction metadata.)
+ // This aspect of a CLOCK cache can make a stressed situation worse by
+ // bogging down the CPU with repeated scans of the cache. And with
+ // strict_capacity_limit=false (normal setting), finding something evictable
+ // doesn't change the outcome of insertion: the entry is inserted anyway
+ // and the cache is allowed to exceed its target capacity if necessary.
+ //
+ // SOLUTION: Eviction is aborted upon seeing some number of pinned
+ // entries before evicting anything, or if the ratio of pinned to evicted
+ // is too high. This setting `eviction_effort_cap` essentially controls both
+ // that allowed initial number of pinned entries and the maximum allowed
+ // ratio. As the pinned size approaches the target cache capacity, roughly
+ // 1/eviction_effort_cap additional portion of the capacity might be kept
+ // in memory and evictable in order to keep CLOCK eviction reasonably
+ // performant. Under the default setting and high stress conditions, this
+ // memory overhead is around 3-5%. Under normal or even moderate stress
+ // conditions, the memory overhead is negligible to zero.
+ //
+ // A large value like 1000 offers some protection with essentially no
+ // memory overhead, while the minimum value of 1 could be useful for a
+ // small cache where roughly doubling in size under stress could be OK to
+ // keep operations very fast.
+ int eviction_effort_cap = 30;
+
HyperClockCacheOptions(
size_t _capacity, size_t _estimated_entry_charge,
int _num_shard_bits = -1, bool _strict_capacity_limit = false,