blob: 2e5eb22f0c4c05adcc53d0feda2ccf5f00e75bc0 [file] [edit]
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
#include "llvm/CAS/UnifiedOnDiskCache.h"
#include "CASTestConfig.h"
#include "OnDiskCommonUtils.h"
#include "llvm/Support/MemoryBuffer.h"
#include "llvm/Support/ThreadPool.h"
#include "llvm/Testing/Support/Error.h"
#include "llvm/Testing/Support/SupportHelpers.h"
#include "gtest/gtest.h"
using namespace llvm;
using namespace llvm::cas;
using namespace llvm::cas::ondisk;
using namespace llvm::unittest::cas;
/// Visits all the files of a directory recursively and returns the sum of their
/// sizes.
static Expected<size_t> countFileSizes(StringRef Path) {
size_t TotalSize = 0;
std::error_code EC;
for (sys::fs::directory_iterator DirI(Path, EC), DirE; !EC && DirI != DirE;
DirI.increment(EC)) {
if (DirI->type() == sys::fs::file_type::directory_file) {
Expected<size_t> Subsize = countFileSizes(DirI->path());
if (!Subsize)
return Subsize.takeError();
TotalSize += *Subsize;
continue;
}
ErrorOr<sys::fs::basic_file_status> Stat = DirI->status();
if (!Stat)
return createFileError(DirI->path(), Stat.getError());
TotalSize += Stat->getSize();
}
if (EC)
return createFileError(Path, EC);
return TotalSize;
}
TEST_P(CustomHasherOnDiskCASTest, UnifiedOnDiskCacheTest) {
auto HashFn = GetParam().HashFn;
StringRef HashName = GetParam().HashName;
size_t HashSize = GetParam().HashSize;
unittest::TempDir Temp("ondisk-unified", /*Unique=*/true);
std::unique_ptr<UnifiedOnDiskCache> UniDB;
const uint64_t SizeLimit = 1024ull * 64;
auto reopenDB = [&]() {
UniDB.reset();
ASSERT_THAT_ERROR(
UnifiedOnDiskCache::open(Temp.path(), SizeLimit, HashName, HashSize)
.moveInto(UniDB),
Succeeded());
};
reopenDB();
HashType RootHash(HashSize);
HashType OtherHash(HashSize);
HashType Key1Hash(HashSize);
HashType Key2Hash(HashSize);
{
OnDiskGraphDB &DB = UniDB->getGraphDB();
std::optional<ObjectID> ID1;
ASSERT_THAT_ERROR(store(DB, "1", {}).moveInto(ID1), Succeeded());
std::optional<ObjectID> ID2;
ASSERT_THAT_ERROR(store(DB, "2", {}).moveInto(ID2), Succeeded());
std::optional<ObjectID> IDRoot;
ASSERT_THAT_ERROR(store(DB, "root", {*ID1, *ID2}).moveInto(IDRoot),
Succeeded());
ArrayRef<uint8_t> Digest = DB.getDigest(*IDRoot);
ASSERT_EQ(Digest.size(), RootHash.size());
llvm::copy(Digest, RootHash.data());
std::optional<ObjectID> IDOther;
ASSERT_THAT_ERROR(store(DB, "other", {}).moveInto(IDOther), Succeeded());
Digest = DB.getDigest(*IDOther);
ASSERT_EQ(Digest.size(), OtherHash.size());
llvm::copy(Digest, OtherHash.data());
Key1Hash = digest("key1");
std::optional<ObjectID> Val;
ASSERT_THAT_ERROR(
cachePut(UniDB->getKeyValueDB(), Key1Hash, *IDRoot).moveInto(Val),
Succeeded());
EXPECT_EQ(IDRoot, Val);
Key2Hash = digest("key2");
std::optional<ObjectID> KeyID;
ASSERT_THAT_ERROR(DB.getReference(Key2Hash).moveInto(KeyID), Succeeded());
ASSERT_THAT_ERROR(cachePut(UniDB->getKeyValueDB(),
UniDB->getGraphDB().getDigest(*KeyID), *ID1)
.moveInto(Val),
Succeeded());
}
auto checkTree = [&](const HashType &Digest, StringRef ExpectedTree) {
OnDiskGraphDB &DB = UniDB->getGraphDB();
std::optional<ObjectID> ID;
ASSERT_THAT_ERROR(DB.getReference(Digest).moveInto(ID), Succeeded());
std::string PrintedTree;
raw_string_ostream OS(PrintedTree);
ASSERT_THAT_ERROR(printTree(DB, *ID, OS), Succeeded());
EXPECT_EQ(PrintedTree, ExpectedTree);
};
auto checkRootTree = [&]() {
return checkTree(RootHash, "root\n 1\n 2\n");
};
auto checkKey = [&](const HashType &Key, StringRef ExpectedData) {
OnDiskGraphDB &DB = UniDB->getGraphDB();
std::optional<ObjectID> Val;
ASSERT_THAT_ERROR(cacheGet(UniDB->getKeyValueDB(), Key).moveInto(Val),
Succeeded());
ASSERT_TRUE(Val.has_value());
std::optional<ondisk::ObjectHandle> Obj;
ASSERT_THAT_ERROR(DB.load(*Val).moveInto(Obj), Succeeded());
EXPECT_EQ(toStringRef(DB.getObjectData(*Obj)), ExpectedData);
};
checkRootTree();
checkTree(OtherHash, "other\n");
checkKey(Key1Hash, "root");
checkKey(Key2Hash, "1");
ASSERT_THAT_ERROR(UniDB->validateActionCache(), Succeeded());
std::optional<ValidationResult> ValidationRes;
ASSERT_THAT_ERROR(UnifiedOnDiskCache::validateIfNeeded(
Temp.path(), HashName, HashSize, /*CheckHash=*/true,
HashFn, /*ForceValidation=*/true)
.moveInto(ValidationRes),
Succeeded());
ASSERT_EQ(ValidationRes, ValidationResult::Valid);
auto storeBigObject = [&](unsigned Index) {
SmallString<1000> Buf;
Buf.append(970, 'a');
raw_svector_ostream(Buf) << Index;
std::optional<ObjectID> ID;
ASSERT_THAT_ERROR(store(UniDB->getGraphDB(), Buf, {}).moveInto(ID),
Succeeded());
};
uint64_t PrevStoreSize = UniDB->getStorageSize();
unsigned Index = 0;
while (!UniDB->hasExceededSizeLimit()) {
storeBigObject(Index++);
}
EXPECT_GT(UniDB->getStorageSize(), PrevStoreSize);
UniDB->setSizeLimit(SizeLimit * 2);
EXPECT_FALSE(UniDB->hasExceededSizeLimit());
UniDB->setSizeLimit(SizeLimit);
EXPECT_TRUE(UniDB->hasExceededSizeLimit());
reopenDB();
EXPECT_FALSE(UniDB->hasExceededSizeLimit());
EXPECT_FALSE(UniDB->needsGarbageCollection());
checkRootTree();
checkKey(Key1Hash, "root");
while (!UniDB->hasExceededSizeLimit()) {
storeBigObject(Index++);
}
PrevStoreSize = UniDB->getStorageSize();
ASSERT_THAT_ERROR(UniDB->close(), Succeeded());
EXPECT_TRUE(UniDB->needsGarbageCollection());
reopenDB();
EXPECT_TRUE(UniDB->needsGarbageCollection());
std::optional<size_t> DirSizeBefore;
ASSERT_THAT_ERROR(countFileSizes(Temp.path()).moveInto(DirSizeBefore),
Succeeded());
ASSERT_THAT_ERROR(UnifiedOnDiskCache::collectGarbage(Temp.path()),
Succeeded());
std::optional<size_t> DirSizeAfter;
ASSERT_THAT_ERROR(countFileSizes(Temp.path()).moveInto(DirSizeAfter),
Succeeded());
EXPECT_LT(*DirSizeAfter, *DirSizeBefore);
reopenDB();
EXPECT_FALSE(UniDB->needsGarbageCollection());
checkRootTree();
checkKey(Key1Hash, "root");
EXPECT_LT(UniDB->getStorageSize(), PrevStoreSize);
// 'Other' tree and 'Key2' got garbage-collected.
{
OnDiskGraphDB &DB = UniDB->getGraphDB();
std::optional<ObjectID> ID;
ASSERT_THAT_ERROR(DB.getReference(OtherHash).moveInto(ID), Succeeded());
EXPECT_FALSE(DB.containsObject(*ID));
std::optional<ObjectID> Val;
ASSERT_THAT_ERROR(cacheGet(UniDB->getKeyValueDB(), Key2Hash).moveInto(Val),
Succeeded());
EXPECT_FALSE(Val.has_value());
}
}
TEST_P(CustomHasherOnDiskCASTest, UnifiedOnDiskCacheConcurrentValidation) {
auto HashFn = GetParam().HashFn;
StringRef HashName = GetParam().HashName;
size_t HashSize = GetParam().HashSize;
auto createCAS = [&](StringRef Path) {
std::unique_ptr<UnifiedOnDiskCache> UniDB;
ASSERT_THAT_ERROR(UnifiedOnDiskCache::open(Path, /*SizeLimit=*/std::nullopt,
HashName, HashSize)
.moveInto(UniDB),
Succeeded());
std::optional<ObjectID> ID;
ASSERT_THAT_ERROR(store(UniDB->getGraphDB(), "1", {}).moveInto(ID),
Succeeded());
};
auto validate = [&](StringRef Path, bool Force) {
return UnifiedOnDiskCache::validateIfNeeded(
Path, HashName, HashSize, /*CheckHash=*/true, HashFn, Force);
};
auto countCorruptDirs = [](StringRef Path) {
unsigned Count = 0;
std::error_code EC;
for (sys::fs::directory_iterator DirI(Path, EC), DirE; !EC && DirI != DirE;
DirI.increment(EC))
if (sys::path::filename(DirI->path()).starts_with("corrupt."))
++Count;
EXPECT_FALSE(EC);
return Count;
};
// Runs \p Fn from multiple threads concurrently, and returns the number of
// times each result occurred. Errors are reported as test failures.
static constexpr unsigned NumTasks = 16;
auto runConcurrently = [&](function_ref<Expected<ValidationResult>()> Fn) {
std::optional<ValidationResult> Results[NumTasks];
std::string Errors[NumTasks];
DefaultThreadPool Pool;
for (unsigned I = 0; I != NumTasks; ++I)
Pool.async([&, I] {
if (Error E = Fn().moveInto(Results[I]))
Errors[I] = toString(std::move(E));
});
Pool.wait();
std::map<ValidationResult, unsigned> Counts;
for (unsigned I = 0; I != NumTasks; ++I) {
EXPECT_EQ(Errors[I], "");
if (Results[I])
++Counts[*Results[I]];
}
return Counts;
};
// Only one of the concurrent validations of valid data is performed, the
// rest see that it has been validated during this boot. Where the boot time
// is not known they are all performed.
const bool BootTimeKnown = isBootTimeKnown();
{
unittest::TempDir Temp("ondisk-unified", /*Unique=*/true);
createCAS(Temp.path());
auto Counts =
runConcurrently([&] { return validate(Temp.path(), /*Force=*/false); });
EXPECT_EQ(Counts[ValidationResult::Valid], BootTimeKnown ? 1u : NumTasks);
EXPECT_EQ(Counts[ValidationResult::Skipped],
BootTimeKnown ? NumTasks - 1 : 0u);
Counts =
runConcurrently([&] { return validate(Temp.path(), /*Force=*/true); });
EXPECT_EQ(Counts[ValidationResult::Valid], NumTasks);
}
// Concurrently validate invalid data and recover if validation fails, as
// done by `llvm-cas -validate-if-needed -allow-recovery`. Exactly one of the
// recoveries is performed and none of them fail, regardless of how the
// validations and recoveries interleave.
for (bool Force : {false, true}) {
SCOPED_TRACE(Force ? "Force" : "NoForce");
unittest::TempDir Temp("ondisk-unified", /*Unique=*/true);
createCAS(Temp.path());
ASSERT_FALSE(sys::fs::remove(Temp.path("v1.1/data.v1")));
auto Counts = runConcurrently([&]() -> Expected<ValidationResult> {
Expected<ValidationResult> Result = validate(Temp.path(), Force);
if (Result)
return Result;
consumeError(Result.takeError());
return UnifiedOnDiskCache::recover(Temp.path());
});
EXPECT_EQ(Counts[ValidationResult::Recovered], 1u);
EXPECT_EQ(Counts[ValidationResult::Recovered] +
Counts[ValidationResult::Skipped] +
Counts[ValidationResult::Valid],
NumTasks);
// The other tasks either skip recovery after a failed validation, or
// validate after the recovery. The latter is skipped unless forced or the
// boot time is not known, in which case the split depends on the
// interleaving.
if (!Force && BootTimeKnown)
EXPECT_EQ(Counts[ValidationResult::Skipped], NumTasks - 1);
EXPECT_EQ(countCorruptDirs(Temp.path()), 1u);
// Recovery counts as validation for this boot.
std::optional<ValidationResult> Result;
ASSERT_THAT_ERROR(validate(Temp.path(), /*Force=*/false).moveInto(Result),
Succeeded());
EXPECT_EQ(Result, BootTimeKnown ? ValidationResult::Skipped
: ValidationResult::Valid);
ASSERT_THAT_ERROR(UnifiedOnDiskCache::recover(Temp.path()).moveInto(Result),
Succeeded());
EXPECT_EQ(Result, ValidationResult::Skipped);
}
}
TEST_P(CustomHasherOnDiskCASTest, UnifiedOnDiskCacheBootTimeMoved) {
if (!isBootTimeKnown())
GTEST_SKIP() << "boot time is not known";
auto HashFn = GetParam().HashFn;
StringRef HashName = GetParam().HashName;
size_t HashSize = GetParam().HashSize;
unittest::TempDir Temp("ondisk-unified", /*Unique=*/true);
{
std::unique_ptr<UnifiedOnDiskCache> UniDB;
ASSERT_THAT_ERROR(UnifiedOnDiskCache::open(Temp.path(),
/*SizeLimit=*/std::nullopt,
HashName, HashSize)
.moveInto(UniDB),
Succeeded());
}
auto validate = [&]() {
return UnifiedOnDiskCache::validateIfNeeded(Temp.path(), HashName, HashSize,
/*CheckHash=*/true, HashFn,
/*ForceValidation=*/false);
};
std::string ValidationPath(Temp.path("v1.validation"));
auto readBootTime = [&]() -> uint64_t {
auto Buf = MemoryBuffer::getFile(ValidationPath);
EXPECT_TRUE(bool(Buf));
uint64_t Value = 0;
if (Buf)
EXPECT_FALSE((*Buf)->getBuffer().trim().getAsInteger(10, Value));
return Value;
};
auto writeBootTime = [&](uint64_t Value) {
std::error_code EC;
raw_fd_ostream OS(ValidationPath, EC);
ASSERT_FALSE(EC);
OS << Value << '\n';
};
std::optional<ValidationResult> Result;
ASSERT_THAT_ERROR(validate().moveInto(Result), Succeeded());
EXPECT_EQ(Result, ValidationResult::Valid);
uint64_t BootTime = readBootTime();
ASSERT_NE(BootTime, 0u);
// The boot time moved back since the validation, e.g. because the clock was
// adjusted, so the recorded one is later. It is still the same boot.
writeBootTime(BootTime + 1);
ASSERT_THAT_ERROR(validate().moveInto(Result), Succeeded());
EXPECT_EQ(Result, ValidationResult::Skipped);
EXPECT_EQ(readBootTime(), BootTime + 1);
// The recorded boot time is earlier, e.g. from an earlier boot, so validation
// is performed and records the current boot time.
writeBootTime(BootTime - 1);
ASSERT_THAT_ERROR(validate().moveInto(Result), Succeeded());
EXPECT_EQ(Result, ValidationResult::Valid);
EXPECT_EQ(readBootTime(), BootTime);
}
TEST_P(CustomHasherOnDiskCASTest, UnifiedOnDiskCacheRepeatedRecovery) {
auto HashFn = GetParam().HashFn;
StringRef HashName = GetParam().HashName;
size_t HashSize = GetParam().HashSize;
unittest::TempDir Temp("ondisk-unified", /*Unique=*/true);
// Each recovery moves the data aside under a new name, even though the
// directory names of earlier recoveries are still taken.
for (unsigned I = 0; I != 3; ++I) {
SCOPED_TRACE(I);
{
std::unique_ptr<UnifiedOnDiskCache> UniDB;
ASSERT_THAT_ERROR(UnifiedOnDiskCache::open(Temp.path(),
/*SizeLimit=*/std::nullopt,
HashName, HashSize)
.moveInto(UniDB),
Succeeded());
std::optional<ObjectID> ID;
ASSERT_THAT_ERROR(store(UniDB->getGraphDB(), "1", {}).moveInto(ID),
Succeeded());
}
ASSERT_FALSE(sys::fs::remove(Temp.path("v1.1/data.v1")));
EXPECT_THAT_EXPECTED(
UnifiedOnDiskCache::validateIfNeeded(Temp.path(), HashName, HashSize,
/*CheckHash=*/true, HashFn,
/*ForceValidation=*/true),
Failed());
std::optional<ValidationResult> Result;
ASSERT_THAT_ERROR(UnifiedOnDiskCache::recover(Temp.path()).moveInto(Result),
Succeeded());
EXPECT_EQ(Result, ValidationResult::Recovered);
EXPECT_TRUE(
sys::fs::exists(Temp.path("corrupt." + std::to_string(I) + ".v1.1")));
EXPECT_FALSE(sys::fs::exists(Temp.path("v1.1")));
}
}