LLVM 24.0.0git
UnifiedOnDiskCache.cpp
Go to the documentation of this file.
1//===----------------------------------------------------------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Encapsulates \p OnDiskGraphDB and \p OnDiskKeyValueDB instances within one
11/// directory while also restricting storage growth with a scheme of chaining
12/// the two most recent directories (primary & upstream), where the primary
13/// "faults-in" data from the upstream one. When the primary (most recent)
14/// directory exceeds its intended limit a new empty directory becomes the
15/// primary one.
16///
17/// Within the top-level directory (the path that \p UnifiedOnDiskCache::open
18/// receives) there are directories named like this:
19///
20/// 'v<version>.<x>'
21/// 'v<version>.<x+1>'
22/// 'v<version>.<x+2>'
23/// ...
24///
25/// 'version' is the version integer for this \p UnifiedOnDiskCache's scheme and
26/// the part after the dot is an increasing integer. The primary directory is
27/// the one with the highest integer and the upstream one is the directory
28/// before it. For example, if the sub-directories contained are:
29///
30/// 'v1.5', 'v1.6', 'v1.7', 'v1.8'
31///
32/// Then the primary one is 'v1.8', the upstream one is 'v1.7', and the rest are
33/// unused directories that can be safely deleted at any time and by any
34/// process.
35///
36/// Contained within the top-level directory is a file named "lock" which is
37/// used for processes to take shared or exclusive locks for the contents of the
38/// top directory. While a \p UnifiedOnDiskCache is open it keeps a shared lock
39/// for the top-level directory; when it closes, if the primary sub-directory
40/// exceeded its limit, it attempts to get an exclusive lock in order to create
41/// a new empty primary directory; if it can't get the exclusive lock it gives
42/// up and lets the next \p UnifiedOnDiskCache instance that closes to attempt
43/// again.
44///
45/// The downside of this scheme is that while \p UnifiedOnDiskCache is open on a
46/// directory, by any process, the storage size in that directory will keep
47/// growing unrestricted. But the major benefit is that garbage-collection can
48/// be triggered on a directory concurrently, at any time and by any process,
49/// without affecting any active readers/writers in the same process or other
50/// processes.
51///
52/// The \c UnifiedOnDiskCache also provides validation and recovery on top of
53/// the underlying on-disk storage. The low-level storage is designed to remain
54/// coherent across regular process crashes, but may be invalid after power loss
55/// or similar system failures. \c UnifiedOnDiskCache::validateIfNeeded allows
56/// validating the contents once per boot (or every time, where the boot time is
57/// not known), and if validation fails (or crashes,
58/// when performed in a separate process) \c UnifiedOnDiskCache::recover can
59/// recover by marking invalid data for garbage collection.
60///
61/// Validation and recovery are serialized by an exclusive lock on the
62/// "v1.validation" file, which records the boot time of the last successful
63/// validation or recovery. Before validating, the file is marked as validation
64/// pending, and the boot time is only written once validation succeeds; a
65/// validation that fails or crashes leaves it pending, so the next validation
66/// is not skipped. Recovery only happens while validation is pending, so when
67/// multiple processes attempt recovery after a failed validation only the first
68/// one recovers.
69///
70/// The data recovery described above requires exclusive access to the CAS, and
71/// it is an error to attempt recovery if the CAS is open in any process/thread.
72/// In order to maximize backwards compatibility with tools that do not perform
73/// validation before opening the CAS, we do not attempt to get exclusive access
74/// until recovery is actually performed, meaning as long as the data is valid
75/// it will not conflict with concurrent use.
76//
77//===----------------------------------------------------------------------===//
78
80#include "OnDiskCommon.h"
81#include "llvm/ADT/STLExtras.h"
82#include "llvm/ADT/ScopeExit.h"
86#include "llvm/ADT/StringRef.h"
91#include "llvm/Support/Errc.h"
92#include "llvm/Support/Error.h"
95#include "llvm/Support/Path.h"
97#include <limits>
98#include <optional>
99
100using namespace llvm;
101using namespace llvm::cas;
102using namespace llvm::cas::ondisk;
103
104/// FIXME: When the version of \p DBDirPrefix is bumped up we need to figure out
105/// how to handle the leftover sub-directories of the previous version, within
106/// the \p UnifiedOnDiskCache::collectGarbage function.
107static constexpr StringLiteral DBDirPrefix = "v1.";
108
109static constexpr StringLiteral ValidationFilename = "v1.validation";
110static constexpr StringLiteral CorruptPrefix = "corrupt.";
111
113 // little endian encoded.
114 assert(Value.size() == sizeof(uint64_t));
116}
117
120 // little endian encoded.
122 static_assert(ValBytes.size() == sizeof(ID.getOpaqueData()));
123 support::endian::write64le(ValBytes.data(), ID.getOpaqueData());
124 return ValBytes;
125}
126
128UnifiedOnDiskCache::faultInFromUpstreamKV(ArrayRef<uint8_t> Key) {
129 assert(UpstreamGraphDB);
130 assert(UpstreamKVDB);
131
132 std::optional<ArrayRef<char>> UpstreamValue;
133 if (Error E = UpstreamKVDB->get(Key).moveInto(UpstreamValue))
134 return std::move(E);
135 if (!UpstreamValue)
136 return std::nullopt;
137
138 // The value is the \p ObjectID in the context of the upstream
139 // \p OnDiskGraphDB instance. Translate it to the context of the primary
140 // \p OnDiskGraphDB instance.
141 ObjectID UpstreamID = getObjectIDFromValue(*UpstreamValue);
142 auto PrimaryID =
143 PrimaryGraphDB->getReference(UpstreamGraphDB->getDigest(UpstreamID));
144 if (LLVM_UNLIKELY(!PrimaryID))
145 return PrimaryID.takeError();
146 return PrimaryKVDB->put(Key, getValueFromObjectID(*PrimaryID));
147}
148
149/// \returns all the 'v<version>.<x>' names of sub-directories, sorted with
150/// ascending order of the integer after the dot. Corrupt directories, if
151/// included, will come first.
153getAllDBDirs(StringRef Path, bool IncludeCorrupt = false) {
154 struct DBDir {
155 uint64_t Order;
156 std::string Name;
157 };
158 SmallVector<DBDir> FoundDBDirs;
159
160 std::error_code EC;
161 for (sys::fs::directory_iterator DirI(Path, EC), DirE; !EC && DirI != DirE;
162 DirI.increment(EC)) {
163 if (DirI->type() != sys::fs::file_type::directory_file)
164 continue;
165 StringRef SubDir = sys::path::filename(DirI->path());
166 if (IncludeCorrupt && SubDir.starts_with(CorruptPrefix)) {
167 FoundDBDirs.push_back({0, std::string(SubDir)});
168 continue;
169 }
170 if (!SubDir.starts_with(DBDirPrefix))
171 continue;
172 uint64_t Order;
173 if (SubDir.substr(DBDirPrefix.size()).getAsInteger(10, Order))
175 "unexpected directory " + DirI->path());
176 FoundDBDirs.push_back({Order, std::string(SubDir)});
177 }
178 if (EC)
179 return createFileError(Path, EC);
180
181 llvm::sort(FoundDBDirs, [](const DBDir &LHS, const DBDir &RHS) -> bool {
182 return LHS.Order < RHS.Order;
183 });
184
186 for (DBDir &Dir : FoundDBDirs)
187 DBDirs.push_back(std::move(Dir.Name));
188 return DBDirs;
189}
190
192 auto DBDirs = getAllDBDirs(Path, /*IncludeCorrupt=*/true);
193 if (!DBDirs)
194 return DBDirs.takeError();
195
196 // FIXME: When the version of \p DBDirPrefix is bumped up we need to figure
197 // out how to handle the leftover sub-directories of the previous version.
198
199 for (unsigned Keep = 2; Keep > 0 && !DBDirs->empty(); --Keep) {
200 StringRef Back(DBDirs->back());
201 if (Back.starts_with(CorruptPrefix))
202 break;
203 DBDirs->pop_back();
204 }
205 return *DBDirs;
206}
207
208/// \returns Given a sub-directory named 'v<version>.<x>', it outputs the
209/// 'v<version>.<x+1>' name.
213 bool Failed = DBDir.substr(DBDirPrefix.size()).getAsInteger(10, Count);
214 assert(!Failed);
215 (void)Failed;
216 OS << DBDirPrefix << Count + 1;
217}
218
222
223static Error validateInProcess(StringRef RootPath, StringRef HashName,
224 unsigned HashByteSize, bool CheckHash,
226 std::shared_ptr<UnifiedOnDiskCache> UniDB;
227 if (Error E = UnifiedOnDiskCache::open(RootPath, std::nullopt, HashName,
228 HashByteSize)
229 .moveInto(UniDB))
230 return E;
231 if (Error E = UniDB->getGraphDB().validate(CheckHash, HashFn))
232 return E;
233 if (Error E = UniDB->validateActionCache())
234 return E;
235 return Error::success();
236}
237
238/// \returns the boot time, or 0 if it is not known, including if getting it
239/// failed. Validation is never skipped where the boot time is not known.
241 static const uint64_t BootTime =
242 expectedToOptional(getBootTime()).value_or(0);
243 return BootTime;
244}
245
246namespace {
247/// The validation file records the state of validation for the data:
248/// - empty: never validated.
249/// - \c ValidationPending: a validation started but did not yet succeed, i.e.
250/// it is in progress, failed, or crashed, and the data has not been
251/// recovered since.
252/// - a boot time: the data was validated or recovered during that boot.
253///
254/// While this object is alive it holds an exclusive lock on the file, which
255/// serializes validation and recovery across processes and threads.
256///
257/// Lock ordering: the validation file lock is always acquired before the
258/// top-level "lock" file, and the latter is only ever acquired exclusively via
259/// a non-blocking try-lock, so that validation and recovery cannot deadlock.
260class LockedValidationFile {
261public:
262 /// Written before validating. It is an integer so that older versions, which
263 /// only know about boot times, still parse the file and treat it as not
264 /// validated. It never matches a boot time.
265 static constexpr uint64_t ValidationPending =
266 std::numeric_limits<uint64_t>::max();
267
268 static Expected<std::unique_ptr<LockedValidationFile>>
269 open(StringRef RootPath) {
270 if (std::error_code EC = sys::fs::create_directories(RootPath))
271 return createFileError(RootPath, EC);
272
273 SmallString<256> PathBuf(RootPath);
275 int FD = -1;
276 if (std::error_code EC = sys::fs::openFileForReadWrite(
278 return createFileError(PathBuf, EC);
279 assert(FD != -1);
280 std::unique_ptr<LockedValidationFile> VF(
281 new LockedValidationFile(PathBuf, FD));
282
283 if (std::error_code EC =
284 lockFileThreadSafe(FD, sys::fs::LockKind::Exclusive))
285 return createFileError(PathBuf, EC);
286 VF->Locked = true;
287
288 SmallString<8> Bytes;
289 if (Error E = sys::fs::readNativeFileToEOF(VF->File, Bytes))
290 return createFileError(PathBuf, std::move(E));
291 if (!Bytes.empty()) {
293 if (StringRef(Bytes).trim().getAsInteger(10, Value))
294 return createFileError(PathBuf, errc::illegal_byte_sequence,
295 "expected integer");
296 VF->State = Value;
297 }
298 return std::move(VF);
299 }
300
301 ~LockedValidationFile() {
302 if (Locked)
304 sys::fs::closeFile(File);
305 }
306
307 /// \returns the boot time of the last successful validation or recovery, or
308 /// 0 if there is none.
309 uint64_t getLastValidBootTime() const {
310 return isValidationPending() ? 0 : State.value_or(0);
311 }
312
313 /// Whether the data was validated or recovered during the boot with
314 /// \p BootTime. Always false where the boot time is not known, i.e. 0,
315 /// since it cannot be told whether that was during the current boot.
316 bool isValidAtBoot(uint64_t BootTime) const {
317 // The boot time can be computed from the current time, so it moves when
318 // the clock is adjusted and the recorded one can be later than BootTime
319 // during the same boot. A reboot always makes it later. Pending is larger
320 // than any boot time, so it needs to be excluded.
321 return BootTime != 0 && !isValidationPending() && State &&
322 BootTime <= *State;
323 }
324
325 bool isValidationPending() const { return State == ValidationPending; }
326
327 Error setValidationPending() { return write(ValidationPending); }
328
329 Error setLastValidBootTime(uint64_t BootTime) { return write(BootTime); }
330
331private:
332 LockedValidationFile(StringRef Path, int FD)
333 : Path(Path), FD(FD), File(sys::fs::convertFDToNativeFile(FD)) {}
334
336 if (State == Value)
337 return Error::success();
338 if (std::error_code EC = sys::fs::resize_file(FD, 0))
339 return createFileError(Path, EC);
340 raw_fd_ostream OS(FD, /*shouldClose=*/false);
341 OS.seek(0); // resize does not reset position
342 OS << Value << '\n';
343 if (OS.has_error())
344 return createFileError(Path, OS.error());
345 State = Value;
346 return Error::success();
347 }
348
349 SmallString<256> Path;
350 int FD;
351 sys::fs::file_t File;
352 bool Locked = false;
353 std::optional<uint64_t> State;
354};
355} // namespace
356
357/// Marks all the database directories in \p RootPath as corrupt, which makes
358/// them eligible for garbage collection. Requires exclusive access to the CAS.
360 SmallString<256> PathBuf(RootPath);
361 sys::path::append(PathBuf, "lock");
362
363 int LockFD = -1;
364 if (std::error_code EC = sys::fs::openFileForReadWrite(
365 PathBuf, LockFD, sys::fs::CD_OpenAlways, sys::fs::OF_None))
366 return createFileError(PathBuf, EC);
368 llvm::scope_exit CloseLock([&]() { sys::fs::closeFile(LockFile); });
369 if (std::error_code EC = tryLockFileThreadSafe(LockFD)) {
370 if (EC == std::errc::no_lock_available)
371 return createFileError(
372 PathBuf, EC,
373 "CAS recovery requires exclusive access but CAS was in use");
374 return createFileError(PathBuf, EC);
375 }
376 llvm::scope_exit UnlockFD([&]() { unlockFileThreadSafe(LockFD); });
377
378 auto DBDirs = getAllDBDirs(RootPath);
379 if (!DBDirs)
380 return DBDirs.takeError();
381
382 for (StringRef DBDir : *DBDirs) {
384 sys::path::append(PathBuf, DBDir);
385 // Pick the first name not taken by earlier recoveries. Checking the error
386 // of the rename is not enough since Windows reports permission denied when
387 // the destination directory exists. The name cannot be taken concurrently
388 // since only garbage collection touches these directories, and it only
389 // removes them.
390 int Attempt = 0, MaxAttempts = 100;
391 SmallString<128> GCPath;
392 for (; Attempt < MaxAttempts; ++Attempt) {
393 GCPath.assign(RootPath);
394 sys::path::append(GCPath,
395 CorruptPrefix + std::to_string(Attempt) + "." + DBDir);
396 if (!sys::fs::exists(GCPath))
397 break;
398 }
399 if (Attempt == MaxAttempts)
400 return createStringError(
402 "rename " + PathBuf +
403 " failed: too many CAS directories awaiting pruning");
404 if (std::error_code EC = sys::fs::rename(PathBuf, GCPath))
405 return createStringError(EC, "rename " + PathBuf + " to " + GCPath +
406 " failed: " + EC.message());
407 }
408 return Error::success();
409}
410
412 StringRef RootPath, StringRef HashName, unsigned HashByteSize,
413 bool CheckHash, OnDiskGraphDB::HashingFuncT HashFn, bool ForceValidation) {
414 std::unique_ptr<LockedValidationFile> VF;
415 if (Error E = LockedValidationFile::open(RootPath).moveInto(VF))
416 return std::move(E);
417
418 std::shared_ptr<ondisk::OnDiskCASLogger> Logger;
419#ifndef _WIN32
420 if (Error E =
421 ondisk::OnDiskCASLogger::openIfEnabled(RootPath).moveInto(Logger))
422 return std::move(E);
423#endif
424
425 uint64_t BootTime = getCachedBootTime();
426 uint64_t ValidationBootTime = VF->getLastValidBootTime();
427
428 bool Skipped = false;
429 std::string LogValidationError;
430
431 llvm::scope_exit Log([&] {
432 if (!Logger)
433 return;
434 Logger->logUnifiedOnDiskCacheValidateIfNeeded(
435 RootPath, BootTime, ValidationBootTime, CheckHash, ForceValidation,
436 LogValidationError, Skipped);
437 });
438
439 if (VF->isValidAtBoot(BootTime) && !ForceValidation) {
440 Skipped = true;
442 }
443
444 // Mark validation as pending until it succeeds, so that a failed or crashed
445 // validation is detected by recovery and by the next validation.
446 if (Error E = VF->setValidationPending())
447 return std::move(E);
448
449 if (Error E = validateInProcess(RootPath, HashName, HashByteSize, CheckHash,
450 HashFn)) {
451 if (Logger)
452 LogValidationError = toStringWithoutConsuming(E);
453 return std::move(E);
454 }
455
456 if (Error E = VF->setLastValidBootTime(BootTime))
457 return std::move(E);
459}
460
462 std::unique_ptr<LockedValidationFile> VF;
463 if (Error E = LockedValidationFile::open(RootPath).moveInto(VF))
464 return std::move(E);
465
466 std::shared_ptr<ondisk::OnDiskCASLogger> Logger;
467#ifndef _WIN32
468 if (Error E =
469 ondisk::OnDiskCASLogger::openIfEnabled(RootPath).moveInto(Logger))
470 return std::move(E);
471#endif
472
473 uint64_t BootTime = getCachedBootTime();
474
475 bool Skipped = false;
476 std::string LogRecoveryError;
477
478 llvm::scope_exit Log([&] {
479 if (!Logger)
480 return;
481 Logger->logUnifiedOnDiskCacheRecover(RootPath, BootTime, LogRecoveryError,
482 Skipped);
483 });
484
485 // Unless validation is still pending, the data has been recovered or
486 // successfully validated since the failed validation, e.g. by a concurrent
487 // process.
488 if (!VF->isValidationPending()) {
489 Skipped = true;
491 }
492
493 if (Error E = markAllDBDirsCorrupt(RootPath)) {
494 if (Logger)
495 LogRecoveryError = toStringWithoutConsuming(E);
496 return std::move(E);
497 }
498
499 if (Error E = VF->setLastValidBootTime(BootTime))
500 return std::move(E);
502}
503
505UnifiedOnDiskCache::open(StringRef RootPath, std::optional<uint64_t> SizeLimit,
506 StringRef HashName, unsigned HashByteSize,
507 OnDiskGraphDB::FaultInPolicy FaultInPolicy) {
508 auto BypassSandbox = sys::sandbox::scopedDisable();
509
510 if (std::error_code EC = sys::fs::create_directories(RootPath))
511 return createFileError(RootPath, EC);
512
513 SmallString<256> PathBuf(RootPath);
514 sys::path::append(PathBuf, "lock");
515 int LockFD = -1;
516 if (std::error_code EC = sys::fs::openFileForReadWrite(
517 PathBuf, LockFD, sys::fs::CD_OpenAlways, sys::fs::OF_None))
518 return createFileError(PathBuf, EC);
519 assert(LockFD != -1);
520 // Locking the directory using shared lock, which will prevent other processes
521 // from creating a new chain (essentially while a \p UnifiedOnDiskCache
522 // instance holds a shared lock the storage for the primary directory will
523 // grow unrestricted).
524 if (std::error_code EC =
526 return createFileError(PathBuf, EC);
527
528 auto DBDirs = getAllDBDirs(RootPath);
529 if (!DBDirs)
530 return DBDirs.takeError();
531 if (DBDirs->empty())
532 DBDirs->push_back((Twine(DBDirPrefix) + "1").str());
533
534 std::shared_ptr<ondisk::OnDiskCASLogger> Logger;
535#ifndef _WIN32
536 if (Error E =
537 ondisk::OnDiskCASLogger::openIfEnabled(RootPath).moveInto(Logger))
538 return std::move(E);
539#endif
540
541 /// If there is only one directory open databases on it. If there are 2 or
542 /// more directories, get the most recent directories and chain them, with the
543 /// most recent being the primary one. The remaining directories are unused
544 /// data than can be garbage-collected.
545 auto UniDB = std::unique_ptr<UnifiedOnDiskCache>(new UnifiedOnDiskCache());
546 std::unique_ptr<OnDiskGraphDB> UpstreamGraphDB;
547 std::unique_ptr<OnDiskKeyValueDB> UpstreamKVDB;
548 if (DBDirs->size() > 1) {
549 StringRef UpstreamDir = *(DBDirs->end() - 2);
550 PathBuf = RootPath;
551 sys::path::append(PathBuf, UpstreamDir);
552 if (Error E =
553 OnDiskGraphDB::open(PathBuf, HashName, HashByteSize,
554 /*UpstreamDB=*/nullptr, Logger, FaultInPolicy)
555 .moveInto(UpstreamGraphDB))
556 return std::move(E);
557 if (Error E = OnDiskKeyValueDB::open(PathBuf, HashName, HashByteSize,
558 /*ValueName=*/"objectid",
559 /*ValueSize=*/sizeof(uint64_t),
560 /*UnifiedCache=*/nullptr, Logger)
561 .moveInto(UpstreamKVDB))
562 return std::move(E);
563 }
564
565 StringRef PrimaryDir = *(DBDirs->end() - 1);
566 PathBuf = RootPath;
567 sys::path::append(PathBuf, PrimaryDir);
568 std::unique_ptr<OnDiskGraphDB> PrimaryGraphDB;
569 if (Error E =
570 OnDiskGraphDB::open(PathBuf, HashName, HashByteSize,
571 UpstreamGraphDB.get(), Logger, FaultInPolicy)
572 .moveInto(PrimaryGraphDB))
573 return std::move(E);
574 std::unique_ptr<OnDiskKeyValueDB> PrimaryKVDB;
575 // \p UnifiedOnDiskCache does manual chaining for key-value requests,
576 // including an extra translation step of the value during fault-in.
577 if (Error E = OnDiskKeyValueDB::open(PathBuf, HashName, HashByteSize,
578 /*ValueName=*/"objectid",
579 /*ValueSize=*/sizeof(uint64_t),
580 UniDB.get(), Logger)
581 .moveInto(PrimaryKVDB))
582 return std::move(E);
583
584 UniDB->RootPath = RootPath;
585 UniDB->SizeLimit = SizeLimit.value_or(0);
586 UniDB->LockFD = LockFD;
587 UniDB->NeedsGarbageCollection = DBDirs->size() > 2;
588 UniDB->PrimaryDBDir = PrimaryDir;
589 UniDB->UpstreamGraphDB = std::move(UpstreamGraphDB);
590 UniDB->PrimaryGraphDB = std::move(PrimaryGraphDB);
591 UniDB->UpstreamKVDB = std::move(UpstreamKVDB);
592 UniDB->PrimaryKVDB = std::move(PrimaryKVDB);
593 UniDB->Logger = std::move(Logger);
594
595 return std::move(UniDB);
596}
597
598void UnifiedOnDiskCache::setSizeLimit(std::optional<uint64_t> SizeLimit) {
599 this->SizeLimit = SizeLimit.value_or(0);
600}
601
603 uint64_t TotalSize = getPrimaryStorageSize();
604 if (UpstreamGraphDB)
605 TotalSize += UpstreamGraphDB->getStorageSize();
606 if (UpstreamKVDB)
607 TotalSize += UpstreamKVDB->getStorageSize();
608 return TotalSize;
609}
610
611uint64_t UnifiedOnDiskCache::getPrimaryStorageSize() const {
612 return PrimaryGraphDB->getStorageSize() + PrimaryKVDB->getStorageSize();
613}
614
616 uint64_t CurSizeLimit = SizeLimit;
617 if (!CurSizeLimit)
618 return false;
619
620 // If the hard limit is beyond 85%, declare above limit and request clean up.
621 unsigned CurrentPercent =
622 std::max(PrimaryGraphDB->getHardStorageLimitUtilization(),
623 PrimaryKVDB->getHardStorageLimitUtilization());
624 if (CurrentPercent > 85)
625 return true;
626
627 // We allow each of the directories in the chain to reach up to half the
628 // intended size limit. Check whether the primary directory has exceeded half
629 // the limit or not, in order to decide whether we need to start a new chain.
630 //
631 // We could check the size limit against the sum of sizes of both the primary
632 // and upstream directories but then if the upstream is significantly larger
633 // than the intended limit, it would trigger a new chain to be created before
634 // the primary has reached its own limit. Essentially in such situation we
635 // prefer reclaiming the storage later in order to have more consistent cache
636 // hits behavior.
637 return (CurSizeLimit / 2) < getPrimaryStorageSize();
638}
639
640Error UnifiedOnDiskCache::close(bool CheckSizeLimit) {
641 auto BypassSandbox = sys::sandbox::scopedDisable();
642
643 if (LockFD == -1)
644 return Error::success(); // already closed.
645 llvm::scope_exit CloseLock([&]() {
646 assert(LockFD >= 0);
648 sys::fs::closeFile(LockFile);
649 LockFD = -1;
650 });
651
652 bool ExceededSizeLimit = CheckSizeLimit ? hasExceededSizeLimit() : false;
653 UpstreamKVDB.reset();
654 PrimaryKVDB.reset();
655 UpstreamGraphDB.reset();
656 PrimaryGraphDB.reset();
657 if (std::error_code EC = unlockFileThreadSafe(LockFD))
658 return createFileError(RootPath, EC);
659
660 if (!ExceededSizeLimit)
661 return Error::success();
662
663 // The primary directory exceeded its intended size limit. Try to get an
664 // exclusive lock in order to create a new primary directory for next time
665 // this \p UnifiedOnDiskCache path is opened.
666
667 if (std::error_code EC = tryLockFileThreadSafe(
668 LockFD, std::chrono::milliseconds(0), sys::fs::LockKind::Exclusive)) {
669 if (EC == errc::no_lock_available)
670 return Error::success(); // couldn't get exclusive lock, give up.
671 return createFileError(RootPath, EC);
672 }
673 llvm::scope_exit UnlockFile([&]() { unlockFileThreadSafe(LockFD); });
674
675 // Managed to get an exclusive lock which means there are no other open
676 // \p UnifiedOnDiskCache instances for the same path, so we can safely start a
677 // new primary directory. To start a new primary directory we just have to
678 // create a new empty directory with the next consecutive index; since this is
679 // an atomic operation we will leave the top-level directory in a consistent
680 // state even if the process dies during this code-path.
681
682 SmallString<256> PathBuf(RootPath);
683 raw_svector_ostream OS(PathBuf);
685 getNextDBDirName(PrimaryDBDir, OS);
686 if (std::error_code EC = sys::fs::create_directory(PathBuf))
687 return createFileError(PathBuf, EC);
688
689 NeedsGarbageCollection = true;
690 return Error::success();
691}
692
693UnifiedOnDiskCache::UnifiedOnDiskCache() = default;
694
696
698 ondisk::OnDiskCASLogger *Logger) {
699 auto DBDirs = getAllGarbageDirs(Path);
700 if (!DBDirs)
701 return DBDirs.takeError();
702
703 SmallString<256> PathBuf(Path);
704 for (StringRef UnusedSubDir : *DBDirs) {
705 sys::path::append(PathBuf, UnusedSubDir);
706 if (Logger)
707 Logger->logUnifiedOnDiskCacheCollectGarbage(PathBuf);
708 if (std::error_code EC = sys::fs::remove_directories(PathBuf))
709 return createFileError(PathBuf, EC);
711 }
712 return Error::success();
713}
714
716 return collectGarbage(RootPath, Logger.get());
717}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
#define LLVM_UNLIKELY(EXPR)
Definition Compiler.h:352
This file declares interface for OnDiskCASLogger, an interface that can be used to log CAS events to ...
This declares OnDiskGraphDB, an ondisk CAS database with a fixed length hash.
This declares OnDiskKeyValueDB, a key value storage database of fixed size key and value.
This file contains some templates that are useful if you are working with the STL at all.
This file defines the scope_exit class, which executes user-defined cleanup logic at scope exit.
This file defines the SmallString class.
This file defines the SmallVector class.
This file contains some functions that are useful when dealing with strings.
static Error validateInProcess(StringRef RootPath, StringRef HashName, unsigned HashByteSize, bool CheckHash, OnDiskGraphDB::HashingFuncT HashFn)
static constexpr StringLiteral DBDirPrefix
FIXME: When the version of DBDirPrefix is bumped up we need to figure out how to handle the leftover ...
static Expected< SmallVector< std::string, 4 > > getAllGarbageDirs(StringRef Path)
static constexpr StringLiteral ValidationFilename
static constexpr StringLiteral CorruptPrefix
static uint64_t getCachedBootTime()
static void getNextDBDirName(StringRef DBDir, llvm::raw_ostream &OS)
static Error markAllDBDirsCorrupt(StringRef RootPath)
Marks all the database directories in RootPath as corrupt, which makes them eligible for garbage coll...
static Expected< SmallVector< std::string, 4 > > getAllDBDirs(StringRef Path, bool IncludeCorrupt=false)
Value * RHS
Value * LHS
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Lightweight error class with error context and mandatory checking.
Definition Error.h:159
static ErrorSuccess success()
Create a success value.
Definition Error.h:336
Tagged union holding either a T or a Error.
Definition Error.h:485
SmallString - A SmallString is just a SmallVector with methods and accessors that make it work better...
Definition SmallString.h:26
void assign(StringRef RHS)
Assign from a StringRef.
Definition SmallString.h:51
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
Definition StringRef.h:597
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM Value Representation.
Definition Value.h:75
Reference to a node.
static ObjectID fromOpaqueData(uint64_t Opaque)
Interface for logging low-level on-disk cas operations.
static LLVM_ABI Expected< std::unique_ptr< OnDiskCASLogger > > openIfEnabled(const Twine &Path)
Create or append to a log file inside the given CAS directory Path if logging is enabled by the envir...
FaultInPolicy
How to fault-in nodes if an upstream database is used.
static LLVM_ABI Expected< std::unique_ptr< OnDiskGraphDB > > open(StringRef Path, StringRef HashName, unsigned HashByteSize, OnDiskGraphDB *UpstreamDB=nullptr, std::shared_ptr< OnDiskCASLogger > Logger=nullptr, FaultInPolicy Policy=FaultInPolicy::FullTree)
Open the on-disk store from a directory.
function_ref< void( ArrayRef< ArrayRef< uint8_t > >, ArrayRef< char >, SmallVectorImpl< uint8_t > &)> HashingFuncT
Hashing function type for validation.
static LLVM_ABI Expected< std::unique_ptr< OnDiskKeyValueDB > > open(StringRef Path, StringRef HashName, unsigned KeySize, StringRef ValueName, size_t ValueSize, UnifiedOnDiskCache *UnifiedCache=nullptr, std::shared_ptr< OnDiskCASLogger > Logger=nullptr)
Open the on-disk store from a directory.
LLVM_ABI Error validate() const
Validate the storage.
static LLVM_ABI ValueBytes getValueFromObjectID(ObjectID ID)
static LLVM_ABI Expected< ValidationResult > recover(StringRef Path)
Recover from invalid data in Path after a failed validateIfNeeded, by marking all the data for garbag...
static LLVM_ABI Expected< std::unique_ptr< UnifiedOnDiskCache > > open(StringRef Path, std::optional< uint64_t > SizeLimit, StringRef HashName, unsigned HashByteSize, OnDiskGraphDB::FaultInPolicy FaultInPolicy=OnDiskGraphDB::FaultInPolicy::FullTree)
Open a UnifiedOnDiskCache instance for a directory.
LLVM_ABI Error close(bool CheckSizeLimit=true)
This is called implicitly at destruction time, so it is not required for a client to call this.
LLVM_ABI Error validateActionCache() const
Validate the action cache only.
static LLVM_ABI ObjectID getObjectIDFromValue(ArrayRef< char > Value)
Helper function to convert the value stored in KeyValueDB and ObjectID.
static LLVM_ABI Expected< ValidationResult > validateIfNeeded(StringRef Path, StringRef HashName, unsigned HashByteSize, bool CheckHash, OnDiskGraphDB::HashingFuncT HashFn, bool ForceValidation)
Validate the data in Path in-process, if it has not been validated since the last system boot.
OnDiskKeyValueDB & getKeyValueDB()
The OnDiskGraphDB instance for the open directory.
std::array< char, sizeof(uint64_t)> ValueBytes
LLVM_ABI Error collectGarbage()
Remove unused data from the current UnifiedOnDiskCache.
LLVM_ABI void setSizeLimit(std::optional< uint64_t > SizeLimit)
Set the size for limiting growth.
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
A raw_ostream that writes to an SmallVector or SmallString.
directory_iterator - Iterates through the entries in path.
std::error_code lockFileThreadSafe(int FD, llvm::sys::fs::LockKind Kind)
Thread-safe alternative to sys::fs::lockFile.
std::error_code unlockFileThreadSafe(int FD)
Thread-safe alternative to sys::fs::unlockFile.
std::error_code tryLockFileThreadSafe(int FD, std::chrono::milliseconds Timeout=std::chrono::milliseconds(0), llvm::sys::fs::LockKind Kind=llvm::sys::fs::LockKind::Exclusive)
Thread-safe alternative to sys::fs::tryLockFile.
LLVM_ABI_FOR_TEST Expected< uint64_t > getBootTime()
Get boot time for the OS.
@ Valid
The data is already valid.
@ Recovered
The data was invalid, but was recovered.
@ Skipped
Validation or recovery was skipped, as it was not needed.
uint64_t read64le(const void *P)
Definition Endian.h:415
void write64le(void *P, uint64_t V)
Definition Endian.h:458
LLVM_ABI std::error_code closeFile(file_t &F)
Close the file object.
std::error_code openFileForReadWrite(const Twine &Name, int &ResultFD, CreationDisposition Disp, OpenFlags Flags, unsigned Mode=0666)
Opens the file with the given name in a write-only or read-write mode, returning its open file descri...
LLVM_ABI std::error_code rename(const Twine &from, const Twine &to)
Rename from to to.
LLVM_ABI Error readNativeFileToEOF(file_t FileHandle, SmallVectorImpl< char > &Buffer, ssize_t ChunkSize=DefaultReadChunkSize)
Reads from FileHandle until EOF, appending to Buffer in chunks of size ChunkSize.
Definition Path.cpp:1227
LLVM_ABI bool exists(const basic_file_status &status)
Does file exist?
Definition Path.cpp:1107
@ CD_OpenAlways
CD_OpenAlways - When opening a file:
Definition FileSystem.h:764
LLVM_ABI std::error_code create_directories(const Twine &path, bool IgnoreExisting=true, perms Perms=owner_all|group_all)
Create all the non-existent directories in path.
Definition Path.cpp:993
LLVM_ABI std::error_code resize_file(int FD, uint64_t Size)
Resize path to size.
LLVM_ABI file_t convertFDToNativeFile(int FD)
Converts from a Posix file descriptor number to a native file handle.
LLVM_ABI std::error_code create_directory(const Twine &path, bool IgnoreExisting=true, perms Perms=owner_all|group_all)
Create the directory in path.
LLVM_ABI std::error_code remove_directories(const Twine &path, bool IgnoreErrors=true)
Recursively delete a directory.
LLVM_ABI StringRef get_separator(Style style=Style::native)
Return the preferred separator for this platform.
Definition Path.cpp:626
LLVM_ABI void remove_filename(SmallVectorImpl< char > &path, Style style=Style::native)
Remove the last component from path unless it is the root dir.
Definition Path.cpp:485
LLVM_ABI StringRef filename(StringRef path LLVM_LIFETIME_BOUND, Style style=Style::native)
Get filename.
Definition Path.cpp:594
LLVM_ABI void append(SmallVectorImpl< char > &path, const Twine &a, const Twine &b="", const Twine &c="", const Twine &d="")
Append to path.
Definition Path.cpp:467
ScopedSetting scopedDisable()
Definition IOSandbox.h:36
This is an optimization pass for GlobalISel generic memory operations.
Error createFileError(const Twine &F, Error E)
Concatenate a source file path and/or name with an Error.
Definition Error.h:1415
LLVM_ABI std::error_code inconvertibleErrorCode()
The value returned by this function can be returned from convertToErrorCode for Error values where no...
Definition Error.cpp:94
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
Error createStringError(std::error_code EC, char const *Fmt, const Ts &... Vals)
Create formatted StringError object.
Definition Error.h:1321
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
@ file_exists
Definition Errc.h:48
@ no_lock_available
Definition Errc.h:61
std::optional< T > expectedToOptional(Expected< T > &&E)
Convert an Expected to an std::optional without doing anything.
Definition Error.h:1117
LLVM_ABI std::string toStringWithoutConsuming(const Error &E)
Like toString(), but does not consume the error.
Definition Error.cpp:81
void sort(IteratorTy Start, IteratorTy End)
Definition STLExtras.h:1652
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
void consumeError(Error Err)
Consume a Error without doing anything.
Definition Error.h:1106
LLVM_ABI Error write(DWPWriter &Out, ArrayRef< std::string > Inputs, OnCuIndexOverflow OverflowOptValue, Dwarf64StrOffsetsPromotion StrOffsetsOptValue, raw_pwrite_stream *OS=nullptr)
Definition DWP.cpp:746
@ Keep
No function return thunk.
Definition CodeGen.h:307
This class wraps the platform specific file handle/descriptor type to provide an unified representati...
Definition File.h:21