Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
53 changes: 53 additions & 0 deletions cachelib/CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -44,6 +44,9 @@ set(CMAKE_POSITION_INDEPENDENT_CODE ON)

option(BUILD_TESTS "If enabled, compile the tests." ON)

option(BUILD_WITH_DTO
"If enabled, build with the DTO library for Intel DSA offload support." OFF)


set(BIN_INSTALL_DIR bin CACHE STRING
"The subdirectory where binaries should be installed")
Expand Down Expand Up @@ -312,6 +315,56 @@ endfunction()
add_thrift_file(OBJECT_CACHE_PERSISTENCE
object_cache/persistence/persistent_data.thrift json)

if (BUILD_WITH_DTO)
# DSA offload runtime (DTO). Resolved here so every cachelib library can
# use it: DTO::dto is linked PUBLIC into cachelib_common (see
# common/CMakeLists.txt), the base library of every other cachelib target,
# which propagates the headers, the link dependency, and the
# CACHELIB_BUILD_WITH_DTO compile definition to the whole tree and to
# consumers of the installed package.
#
# The DSA checksum offload code requires a DTO with the caller-allocated
# async submit/poll API and the raw-CRC32C convention, which upstream
# intel/DTO does not have — so when no installed DTO cmake package is
# found, fetch and build the pinned revision in-tree rather than linking
# whatever libdto happens to be on the system.
set(CACHELIB_DTO_GIT_REPOSITORY "https://github.com/byrnedj/DTO.git"
CACHE STRING "Git repository for DTO when no installed copy is found")
set(CACHELIB_DTO_GIT_TAG "991f4f9084095af78e2d7ca2dcb9a8c6069f8904"
CACHE STRING "DTO revision to fetch when no installed copy is found")

find_package(DTO CONFIG QUIET)
if (DTO_FOUND)
message(STATUS "Using installed DTO package: ${DTO_DIR}")
else()
if (CMAKE_VERSION VERSION_LESS 3.14)
message(FATAL_ERROR
"BUILD_WITH_DTO needs an installed DTO cmake package or "
"CMake >= 3.14 to fetch one. Install DTO from "
"${CACHELIB_DTO_GIT_REPOSITORY} at ${CACHELIB_DTO_GIT_TAG}, "
"or upgrade CMake.")
endif()
# DTO links against libaccel-config and libnuma; check for them here so
# a missing system package fails at configure time with a clear message
# instead of at link time inside the fetched project.
find_library(ACCEL_CONFIG_LIBRARY accel-config)
find_library(NUMA_LIBRARY numa)
if (NOT ACCEL_CONFIG_LIBRARY OR NOT NUMA_LIBRARY)
message(FATAL_ERROR
"BUILD_WITH_DTO requires the libaccel-config and libnuma "
"development packages (e.g. apt install libaccel-config-dev "
"libnuma-dev)")
endif()
message(STATUS "DTO not installed; fetching "
"${CACHELIB_DTO_GIT_REPOSITORY} @ ${CACHELIB_DTO_GIT_TAG}")
include(FetchContent)
FetchContent_Declare(dto
GIT_REPOSITORY "${CACHELIB_DTO_GIT_REPOSITORY}"
GIT_TAG "${CACHELIB_DTO_GIT_TAG}")
FetchContent_MakeAvailable(dto)
endif()
endif()

add_subdirectory (common)
add_subdirectory (shm)
add_subdirectory (navy)
Expand Down
6 changes: 6 additions & 0 deletions cachelib/allocator/Cache.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -426,6 +426,12 @@ void CacheBase::updateGlobalCacheStats(const std::string& statPrefix) const {
if (stats.nvmCacheEnabled) {
counters_.updateDelta(statPrefix + "nvm.alloc_attempts",
stats.numNvmAllocAttempts);
counters_.updateDelta(statPrefix + "nvm.copy_out_offloaded",
stats.numNvmCopyOutOffloaded);
counters_.updateDelta(statPrefix + "nvm.copy_out_offloaded_bytes",
stats.numNvmCopyOutOffloadedBytes);
counters_.updateDelta(statPrefix + "nvm.copy_out_fallbacks",
stats.numNvmCopyOutFallbacks);
counters_.updateDelta(statPrefix + "nvm.destructor_alloc",
stats.numNvmAllocForItemDestructor);
counters_.updateDelta(statPrefix + "nvm.destructor_alloc_errors",
Expand Down
7 changes: 5 additions & 2 deletions cachelib/allocator/CacheStats.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -54,10 +54,10 @@ struct SizeVerify {};
void Stats::populateGlobalCacheStats(GlobalCacheStats& ret) const {
#ifndef SKIP_SIZE_VERIFY
#ifdef __GLIBCXX__
#define EXPECTED_SIZE 16944
#define EXPECTED_SIZE 17200
#endif
#ifdef _LIBCPP_VERSION
#define EXPECTED_SIZE 16944
#define EXPECTED_SIZE 17200
#endif
SizeVerify<sizeof(Stats)> a = SizeVerify<EXPECTED_SIZE>{};
std::ignore = a;
Expand Down Expand Up @@ -106,6 +106,9 @@ void Stats::populateGlobalCacheStats(GlobalCacheStats& ret) const {
ret.numInsertOrReplaceInserted = numInsertOrReplaceInserted.get();
ret.numInsertOrReplaceReplaced = numInsertOrReplaceReplaced.get();
ret.numNvmAllocAttempts = numNvmAllocAttempts.get();
ret.numNvmCopyOutOffloaded = numNvmCopyOutOffloaded.get();
ret.numNvmCopyOutOffloadedBytes = numNvmCopyOutOffloadedBytes.get();
ret.numNvmCopyOutFallbacks = numNvmCopyOutFallbacks.get();
ret.numNvmAllocForItemDestructor = numNvmAllocForItemDestructor.get();
ret.numNvmItemDestructorAllocErrors = numNvmItemDestructorAllocErrors.get();

Expand Down
5 changes: 5 additions & 0 deletions cachelib/allocator/CacheStats.h
Original file line number Diff line number Diff line change
Expand Up @@ -456,6 +456,11 @@ struct GlobalCacheStats {
// attempts made from nvm cache to allocate an item for promotion
uint64_t numNvmAllocAttempts{0};

// flash-hit copy-outs performed by Intel DSA, bytes moved, CPU fallbacks
uint64_t numNvmCopyOutOffloaded{0};
uint64_t numNvmCopyOutOffloadedBytes{0};
uint64_t numNvmCopyOutFallbacks{0};

// attempts made from nvm cache to allocate an item for its destructor
uint64_t numNvmAllocForItemDestructor{0};
// heap allocate errors for item destructor
Expand Down
6 changes: 6 additions & 0 deletions cachelib/allocator/CacheStatsInternal.h
Original file line number Diff line number Diff line change
Expand Up @@ -141,6 +141,12 @@ struct Stats {
// attempts made from nvm cache to allocate an item for promotion
TLCounter numNvmAllocAttempts{0};

// flash-hit copy-outs (Navy buffer -> DRAM item) performed by Intel DSA,
// the bytes they moved, and offload attempts that fell back to the CPU
TLCounter numNvmCopyOutOffloaded{0};
TLCounter numNvmCopyOutOffloadedBytes{0};
TLCounter numNvmCopyOutFallbacks{0};

// attempts made from nvm cache to allocate an item for its destructor
TLCounter numNvmAllocForItemDestructor{0};
// heap allocate errors for item destructor
Expand Down
6 changes: 6 additions & 0 deletions cachelib/allocator/nvmcache/NavyConfig.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -323,6 +323,10 @@ std::map<std::string, std::string> EnginesConfig::serialize() const {
folly::to<std::string>(blockCache().getNumInMemBuffers());
configMap["navyConfig::blockCacheDataChecksum"] =
blockCache().getDataChecksum() ? "true" : "false";
configMap["navyConfig::blockCacheChecksumOffload"] =
blockCache().getChecksumOffload() ? "true" : "false";
configMap["navyConfig::blockCacheChecksumOffloadMinSize"] =
folly::to<std::string>(blockCache().getChecksumOffloadMinSize());
configMap["navyConfig::blockCacheSegmentedFifoSegmentRatio"] =
folly::join(",", blockCache().getSFifoSegmentRatio());

Expand All @@ -335,6 +339,8 @@ std::map<std::string, std::string> EnginesConfig::serialize() const {
folly::to<std::string>(bigHash().getBucketBfSize());
configMap["navyConfig::bigHashSmallItemMaxSize"] =
folly::to<std::string>(bigHash().getSmallItemMaxSize());
configMap["navyConfig::bigHashChecksumOffload"] =
bigHash().getChecksumOffload() ? "true" : "false";
return configMap;
}

Expand Down
75 changes: 75 additions & 0 deletions cachelib/allocator/nvmcache/NavyConfig.h
Original file line number Diff line number Diff line change
Expand Up @@ -536,6 +536,36 @@ class BlockCacheConfig {
return *this;
}

// Offload data checksumming (fused with the value copy on the write path)
// to Intel DSA via the DTO library. Requires data checksum to be enabled
// and CacheLib built with BUILD_WITH_DTO. @minSize is the minimum value
// size to use the offloaded path; smaller values are checksummed in
// software, since submitting and polling a descriptor costs about what the
// CPU needs to CRC 16-32 KiB (16 KiB measured as the break-even).
BlockCacheConfig& setChecksumOffload(bool checksumOffload,
uint32_t minSize = 16384) noexcept {
checksumOffload_ = checksumOffload;
checksumOffloadMinSize_ = minSize;
return *this;
}

// Separate minimum size for offloading checksum VERIFICATION on the read,
// reclaim and reinsertion paths; 0 = same as the write-path gate. Verifying
// has no CPU work to overlap with the accelerator, so its break-even is
// higher than the fused write.
BlockCacheConfig& setChecksumOffloadReadMinSize(uint32_t minSize) noexcept {
checksumOffloadReadMinSize_ = minSize;
return *this;
}

// Cache-control hint on the fused write-path copy (default true): steer the
// value bytes toward the CPU cache. Turn off when the next reader of the
// region buffer is the device (directFlush / flush copy offload).
BlockCacheConfig& setChecksumOffloadCacheControl(bool enable) noexcept {
checksumOffloadCacheControl_ = enable;
return *this;
}

BlockCacheConfig& setPreciseRemove(bool preciseRemove) noexcept {
preciseRemove_ = preciseRemove;
return *this;
Expand All @@ -561,6 +591,17 @@ class BlockCacheConfig {
return *this;
}

// When directFlush is off: copy the region buffer into the flush write
// buffer with one Intel DSA Memory Move instead of memcpy. Requires
// CacheLib built with BUILD_WITH_DTO; falls back to memcpy if DSA is
// unusable. The write buffers are pooled, so after the first flush per
// buffer their pages are resident; a work queue with block-on-fault covers
// the first touch.
BlockCacheConfig& setFlushCopyOffload(bool enable) noexcept {
flushCopyOffload_ = enable;
return *this;
}

BlockCacheConfig& setAllocatorCount(uint32_t numAllocators) noexcept {
allocatorsPerPriority_ = {numAllocators};
return *this;
Expand Down Expand Up @@ -617,13 +658,26 @@ class BlockCacheConfig {

bool getDataChecksum() const { return dataChecksum_; }

bool getChecksumOffload() const { return checksumOffload_; }

uint32_t getChecksumOffloadMinSize() const { return checksumOffloadMinSize_; }

uint32_t getChecksumOffloadReadMinSize() const {
return checksumOffloadReadMinSize_;
}

bool getChecksumOffloadCacheControl() const {
return checksumOffloadCacheControl_;
}

uint64_t getSize() const { return size_; }

bool isRegionManagerFlushAsync() const { return regionManagerFlushAsync_; }

bool isRecoverEvictionPolicy() const { return recoverEvictionPolicy_; }

bool isDirectFlush() const { return directFlush_; }
bool isFlushCopyOffload() const { return flushCopyOffload_; }

bool isCombinedEntryBlockEnabled() const { return useCombinedEntryBlock_; }

Expand Down Expand Up @@ -660,6 +714,12 @@ class BlockCacheConfig {
uint32_t regionSize_{16 * 1024 * 1024};
// Whether enabling data checksum for Navy BlockCache.
bool dataChecksum_{true};
// Whether to offload data checksumming to Intel DSA (fused with the value
// copy on the write path), and the minimum value size to do so.
bool checksumOffload_{false};
uint32_t checksumOffloadMinSize_{16384};
uint32_t checksumOffloadReadMinSize_{0};
bool checksumOffloadCacheControl_{true};
// Whether to remove an item by checking the key (true) or only the hash value
// (false).
bool preciseRemove_{false};
Expand All @@ -677,6 +737,9 @@ class BlockCacheConfig {
// Whether to write region buffer directly without intermediate copy.
bool directFlush_{false};

// Whether to do the flush copy (when not directFlush) on Intel DSA.
bool flushCopyOffload_{false};

// Whether to use Combined entry block (For index entries and small sized
// items).
// Only FixedSizeIndex will support this and it doesn't work with
Expand Down Expand Up @@ -739,6 +802,14 @@ class BigHashConfig {
return *this;
}

// Offload bucket checksumming to Intel DSA via the DTO library. Most
// beneficial with large buckets (16KB+). Requires CacheLib built with
// BUILD_WITH_DTO.
BigHashConfig& setChecksumOffload(bool checksumOffload) noexcept {
checksumOffload_ = checksumOffload;
return *this;
}

bool isBloomFilterEnabled() const { return bucketBfSize_ > 0; }

unsigned int getSizePct() const { return sizePct_; }
Expand All @@ -751,6 +822,8 @@ class BigHashConfig {

uint8_t getNumMutexesPower() const { return numMutexesPower_; }

bool getChecksumOffload() const { return checksumOffload_; }

private:
// Percentage of how much of the device out of all is given to BigHash
// engine in Navy, e.g. 50.
Expand All @@ -766,6 +839,8 @@ class BigHashConfig {
uint64_t smallItemMaxSize_{};
// numMutexes = 1 << numMutexesPower_.
uint8_t numMutexesPower_{14};
// Whether to offload bucket checksumming to Intel DSA.
bool checksumOffload_{false};
};

// Config for a pair of small,large engines.
Expand Down
9 changes: 9 additions & 0 deletions cachelib/allocator/nvmcache/NavySetup.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -140,6 +140,8 @@ uint64_t setupBigHash(const navy::BigHashConfig& bigHashConfig,
// Set number of mutexes from config
bigHash->setNumMutexesPower(bigHashConfig.getNumMutexesPower());

bigHash->setChecksumOffload(bigHashConfig.getChecksumOffload());

proto.setBigHash(std::move(bigHash), bigHashConfig.getSmallItemMaxSize());

if (bigHashCacheOffset <= bigHashStartOffsetLimit) {
Expand Down Expand Up @@ -196,6 +198,12 @@ uint64_t setupBlockCache(const navy::BlockCacheConfig& blockCacheConfig,
auto blockCache = cachelib::navy::createBlockCacheProto();
blockCache->setLayout(blockCacheOffset, blockCacheSize, regionSize);
blockCache->setChecksum(blockCacheConfig.getDataChecksum());
blockCache->setChecksumOffload(blockCacheConfig.getChecksumOffload(),
blockCacheConfig.getChecksumOffloadMinSize());
blockCache->setChecksumOffloadReadMinSize(
blockCacheConfig.getChecksumOffloadReadMinSize());
blockCache->setChecksumOffloadCacheControl(
blockCacheConfig.getChecksumOffloadCacheControl());

// set eviction policy
auto segmentRatio = blockCacheConfig.getSFifoSegmentRatio();
Expand All @@ -214,6 +222,7 @@ uint64_t setupBlockCache(const navy::BlockCacheConfig& blockCacheConfig,
blockCache->setRecoverEvictionPolicy(
blockCacheConfig.isRecoverEvictionPolicy());
blockCache->setDirectFlush(blockCacheConfig.isDirectFlush());
blockCache->setFlushCopyOffload(blockCacheConfig.isFlushCopyOffload());
blockCache->setUseCombinedEntryBlock(
blockCacheConfig.isCombinedEntryBlockEnabled());
blockCache->setNumAllocatorsPerPriority(
Expand Down
Loading
Loading