1//===-- secondary.h ---------------------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#ifndef SCUDO_SECONDARY_H_
10#define SCUDO_SECONDARY_H_
11
12#ifndef __STDC_FORMAT_MACROS
13// Ensure PRId64 macro is available
14#define __STDC_FORMAT_MACROS 1
15#endif
16#include <inttypes.h>
17
18#include "chunk.h"
19#include "common.h"
20#include "list.h"
21#include "mem_map.h"
22#include "memtag.h"
23#include "mutex.h"
24#include "options.h"
25#include "stats.h"
26#include "string_utils.h"
27#include "thread_annotations.h"
28#include "tracing.h"
29#include "vector.h"
30
31namespace scudo {
32
33// This allocator wraps the platform allocation primitives, and as such is on
34// the slower side and should preferably be used for larger sized allocations.
35// Blocks allocated will be preceded and followed by a guard page, and hold
36// their own header that is not checksummed: the guard pages and the Combined
37// header should be enough for our purpose.
38
39namespace LargeBlock {
40
41struct alignas(Max<uptr>(A: archSupportsMemoryTagging()
42 ? archMemoryTagGranuleSize()
43 : 1,
44 B: 1U << SCUDO_MIN_ALIGNMENT_LOG)) Header {
45 LargeBlock::Header *Prev;
46 LargeBlock::Header *Next;
47 uptr CommitBase;
48 uptr CommitSize;
49 MemMapT MemMap;
50};
51
52static_assert(sizeof(Header) % (1U << SCUDO_MIN_ALIGNMENT_LOG) == 0, "");
53static_assert(!archSupportsMemoryTagging() ||
54 sizeof(Header) % archMemoryTagGranuleSize() == 0,
55 "");
56
57constexpr uptr getHeaderSize() { return sizeof(Header); }
58
59template <typename Config> uptr addHeaderTag(uptr Ptr) {
60 if (allocatorSupportsMemoryTagging<Config>())
61 return addFixedTag(Ptr, Tag: 1);
62 return Ptr;
63}
64
65template <typename Config> Header *getHeader(uptr Ptr) {
66 return reinterpret_cast<Header *>(addHeaderTag<Config>(Ptr)) - 1;
67}
68
69template <typename Config> Header *getHeader(const void *Ptr) {
70 return getHeader<Config>(reinterpret_cast<uptr>(Ptr));
71}
72
73} // namespace LargeBlock
74
75static inline void unmap(MemMapT &MemMap) { MemMap.unmap(); }
76
77namespace {
78
79struct CachedBlock {
80 static constexpr u16 CacheIndexMax = UINT16_MAX;
81 static constexpr u16 EndOfListVal = CacheIndexMax;
82
83 // We allow a certain amount of fragmentation and part of the fragmented bytes
84 // will be released by `releaseAndZeroPagesToOS()`. This increases the chance
85 // of cache hit rate and reduces the overhead to the RSS at the same time. See
86 // more details in the `MapAllocatorCache::retrieve()` section.
87 //
88 // We arrived at this default value after noticing that mapping in larger
89 // memory regions performs better than releasing memory and forcing a cache
90 // hit. According to the data, it suggests that beyond 4 pages, the release
91 // execution time is longer than the map execution time. In this way,
92 // the default is dependent on the platform.
93 static constexpr uptr MaxReleasedCachePages = 4U;
94
95 uptr CommitBase = 0;
96 uptr CommitSize = 0;
97 uptr BlockBegin = 0;
98 MemMapT MemMap = {};
99 u64 Time = 0;
100 u16 Next = 0;
101 u16 Prev = 0;
102
103 enum CacheFlags : u16 {
104 None = 0,
105 NoAccess = 0x1,
106 };
107 CacheFlags Flags = CachedBlock::None;
108
109 bool isValid() { return CommitBase != 0; }
110
111 void invalidate() { CommitBase = 0; }
112};
113} // namespace
114
115template <typename Config> class MapAllocatorNoCache {
116public:
117 void init(UNUSED s32 ReleaseToOsInterval) {}
118 CachedBlock retrieve(UNUSED uptr MaxAllowedFragmentedBytes, UNUSED uptr Size,
119 UNUSED uptr Alignment, UNUSED uptr HeadersSize,
120 UNUSED uptr &EntryHeaderPos) {
121 return {};
122 }
123 void store(UNUSED Options Options, UNUSED uptr CommitBase,
124 UNUSED uptr CommitSize, UNUSED uptr BlockBegin,
125 UNUSED MemMapT MemMap) {
126 // This should never be called since canCache always returns false.
127 UNREACHABLE(
128 "It is not valid to call store on MapAllocatorNoCache objects.");
129 }
130
131 bool canCache(UNUSED uptr Size) { return false; }
132 void disable() {}
133 void enable() {}
134 void releaseToOS(ReleaseToOS) {}
135 void disableMemoryTagging() {}
136 void unmapTestOnly() {}
137 bool setOption(Option O, UNUSED sptr Value) {
138 if (O == Option::ReleaseInterval || O == Option::MaxCacheEntriesCount ||
139 O == Option::MaxCacheEntrySize || O == Option::MaxCacheResidentBytes)
140 return false;
141 // Not supported by the Secondary Cache, but not an error either.
142 return true;
143 }
144
145 uptr getMaxResidentBytesTestOnly() const { return 0; }
146 uptr getCurrentResidentBytesTestOnly() const { return 0; }
147
148 void getStats(UNUSED ScopedString *Str) {
149 Str->append(Format: "Secondary Cache Disabled\n");
150 }
151};
152
153static const uptr MaxUnreleasedCachePages = 4U;
154
155template <typename Config>
156bool mapSecondary(const Options &Options, uptr CommitBase, uptr CommitSize,
157 uptr AllocPos, uptr Flags, MemMapT &MemMap) {
158 Flags |= MAP_RESIZABLE;
159 Flags |= MAP_ALLOWNOMEM;
160
161 const uptr PageSize = getPageSizeCached();
162 if (SCUDO_TRUSTY) {
163 /*
164 * On Trusty we need AllocPos to be usable for shared memory, which cannot
165 * cross multiple mappings. This means we need to split around AllocPos
166 * and not over it. We can only do this if the address is page-aligned.
167 */
168 const uptr TaggedSize = AllocPos - CommitBase;
169 if (useMemoryTagging<Config>(Options) && isAligned(X: TaggedSize, Alignment: PageSize)) {
170 DCHECK_GT(TaggedSize, 0);
171 return MemMap.remap(Addr: CommitBase, Size: TaggedSize, Name: "scudo:secondary",
172 MAP_MEMTAG | Flags) &&
173 MemMap.remap(Addr: AllocPos, Size: CommitSize - TaggedSize, Name: "scudo:secondary",
174 Flags);
175 } else {
176 const uptr RemapFlags =
177 (useMemoryTagging<Config>(Options) ? MAP_MEMTAG : 0) | Flags;
178 return MemMap.remap(Addr: CommitBase, Size: CommitSize, Name: "scudo:secondary",
179 Flags: RemapFlags);
180 }
181 }
182
183 // AllocPos is assumed to be page-aligned when memory tagging is enabled.
184 // Therefore the page right before AllocPos is MTE-tagged and the header
185 // resides in this MTE-tagged page.
186 if (useMemoryTagging<Config>(Options)) {
187 const uptr PageSize = getPageSizeCached();
188 const uptr MteStart = AllocPos - PageSize;
189
190 DCHECK(AllocPos % PageSize == 0U);
191 DCHECK(MteStart % PageSize == 0U);
192
193 DCHECK_GE(MteStart, CommitBase);
194 DCHECK_LE(AllocPos, CommitBase + CommitSize);
195 return MemMap.remap(Addr: MteStart, Size: PageSize, Name: "scudo:secondary",
196 MAP_MEMTAG | Flags) &&
197 MemMap.remap(Addr: AllocPos, Size: CommitBase + CommitSize - AllocPos,
198 Name: "scudo:secondary", Flags);
199 } else {
200 const uptr RemapFlags =
201 (useMemoryTagging<Config>(Options) ? MAP_MEMTAG : 0) | Flags;
202 return MemMap.remap(Addr: CommitBase, Size: CommitSize, Name: "scudo:secondary", Flags: RemapFlags);
203 }
204}
205
206// Template specialization to avoid producing zero-length array
207template <typename T, size_t Size> class NonZeroLengthArray {
208public:
209 T &operator[](uptr Idx) { return values[Idx]; }
210
211private:
212 T values[Size];
213};
214template <typename T> class NonZeroLengthArray<T, 0> {
215public:
216 T &operator[](uptr UNUSED Idx) { UNREACHABLE("Unsupported!"); }
217};
218
219// The default unmap callback is simply scudo::unmap.
220// In testing, a different unmap callback is used to
221// record information about unmaps in the cache
222template <typename Config, void (*unmapCallBack)(MemMapT &) = unmap>
223class MapAllocatorCache {
224public:
225 void getStats(ScopedString *Str) {
226 ScopedLock L(Mutex);
227 Str->append(Format: "Config Stats Secondary: ");
228 Config::getConfigValues(Str);
229 uptr Integral;
230 uptr Fractional;
231 computePercentage(Numerator: SuccessfulRetrieves, Denominator: CallsToRetrieve, Integral: &Integral,
232 Fractional: &Fractional);
233 const s32 Interval = atomic_load_relaxed(A: &ReleaseToOsIntervalMs);
234 Str->append(
235 Format: "Stats: MapAllocatorCache: EntriesCount: %zu, "
236 "MaxEntriesCount: %u, MaxEntrySize: %zu, ReleaseToOsSkips: "
237 "%zu, ReleaseToOsIntervalMs = %d, Unmapped due to eviction: %u, "
238 "MaxResidentBytes: %zu, CurrentResidentBytes: %zu\n",
239 LRUEntries.size(), atomic_load_relaxed(A: &MaxEntriesCount),
240 atomic_load_relaxed(A: &MaxEntrySize),
241 atomic_load_relaxed(A: &ReleaseToOsSkips), Interval >= 0 ? Interval : -1,
242 EvictedCount, MaxResidentBytes, CurrentResidentBytes);
243 Str->append(Format: "Stats: CacheRetrievalStats: SuccessRate: %u/%u "
244 "(%zu.%02zu%%)\n",
245 SuccessfulRetrieves, CallsToRetrieve, Integral, Fractional);
246 Str->append(Format: "Cache Entry Info (Most Recent -> Least Recent):\n");
247
248 for (CachedBlock &Entry : LRUEntries) {
249 Str->append(Format: " StartBlockAddress: 0x%zx, EndBlockAddress: 0x%zx, "
250 "BlockSize: %zu%s, Flags: %s",
251 Entry.CommitBase, Entry.CommitBase + Entry.CommitSize,
252 Entry.CommitSize, Entry.Time == 0 ? " [R]" : "",
253 Entry.Flags & CachedBlock::NoAccess ? "NoAccess" : "None");
254 const s64 ResidentPages =
255 Entry.MemMap.getResidentPages(From: Entry.CommitBase, Size: Entry.CommitSize);
256
257 if (ResidentPages >= 0) {
258 Str->append(Format: ", Resident Pages: %" PRId64 "/%zu", ResidentPages,
259 Entry.CommitSize / getPageSizeCached());
260 }
261 Str->append(Format: "\n");
262 }
263 }
264
265 // Ensure the default maximum specified fits the array.
266 static_assert(Config::getDefaultMaxEntriesCount() <=
267 Config::getEntriesArraySize(),
268 "");
269 // Ensure the cache entry array size fits in the LRU list Next and Prev
270 // index fields
271 static_assert(Config::getEntriesArraySize() <= CachedBlock::CacheIndexMax,
272 "Cache entry array is too large to be indexed.");
273
274 void init(s32 ReleaseToOsInterval) NO_THREAD_SAFETY_ANALYSIS {
275 DCHECK_EQ(LRUEntries.size(), 0U);
276 setOption(O: Option::MaxCacheEntriesCount,
277 Value: static_cast<sptr>(Config::getDefaultMaxEntriesCount()));
278 setOption(O: Option::MaxCacheEntrySize,
279 Value: static_cast<sptr>(Config::getDefaultMaxEntrySize()));
280 setOption(O: Option::MaxCacheResidentBytes,
281 Value: static_cast<sptr>(Config::getDefaultMaxCacheResidentBytes()));
282 // The default value in the cache config has the higher priority.
283 if (Config::getDefaultReleaseToOsIntervalMs() != INT32_MIN)
284 ReleaseToOsInterval = Config::getDefaultReleaseToOsIntervalMs();
285 setOption(O: Option::ReleaseInterval, Value: static_cast<sptr>(ReleaseToOsInterval));
286
287 LRUEntries.clear();
288 LRUEntries.init(Base: Entries, BaseSize: sizeof(Entries));
289 OldestPresentEntry = nullptr;
290
291 AvailEntries.clear();
292 AvailEntries.init(Base: Entries, BaseSize: sizeof(Entries));
293 for (u32 I = 0; I < Config::getEntriesArraySize(); I++)
294 AvailEntries.push_back(X: &Entries[I]);
295 }
296
297 void store(const Options &Options, uptr CommitBase, uptr CommitSize,
298 uptr BlockBegin, MemMapT MemMap) EXCLUDES(Mutex) {
299 DCHECK(canCache(CommitSize));
300
301 const s32 Interval = atomic_load_relaxed(A: &ReleaseToOsIntervalMs);
302 u64 Time;
303 CachedBlock Entry;
304
305 Entry.CommitBase = CommitBase;
306 Entry.CommitSize = CommitSize;
307 Entry.BlockBegin = BlockBegin;
308 Entry.MemMap = MemMap;
309 Entry.Time = UINT64_MAX;
310 Entry.Flags = CachedBlock::None;
311
312 bool MemoryTaggingEnabled = useMemoryTagging<Config>(Options);
313 if (MemoryTaggingEnabled) {
314 if (Interval == 0 && !SCUDO_FUCHSIA) {
315 Entry.Time = 0;
316 Entry.MemMap.releaseAndZeroPagesToOS(From: Entry.CommitBase,
317 Size: Entry.CommitSize);
318 }
319 // MAP_NOACCESS or PROT_NONE does not strip PROT_MTE.
320 Entry.MemMap.setMemoryPermission(Addr: Entry.CommitBase, Size: Entry.CommitSize,
321 MAP_NOACCESS);
322 Entry.Flags = CachedBlock::NoAccess;
323 }
324
325 // Usually only one entry will be evicted from the cache.
326 // Only in the rare event that the cache shrinks in real-time
327 // due to a decrease in the configurable value MaxEntriesCount
328 // will more than one cache entry be evicted.
329 // The vector is used to save the MemMaps of evicted entries so
330 // that the unmap call can be performed outside the lock
331 Vector<MemMapT, 1U> EvictionMemMaps;
332
333 do {
334 ScopedLock L(Mutex);
335
336 // Time must be computed under the lock to ensure
337 // that the LRU cache remains sorted with respect to
338 // time in a multithreaded environment
339 Time = getMonotonicTimeFast();
340 if (Entry.Time != 0)
341 Entry.Time = Time;
342
343 if (MemoryTaggingEnabled && !useMemoryTagging<Config>(Options)) {
344 // If we get here then memory tagging was disabled in between when we
345 // read Options and when we locked Mutex. We can't insert our entry into
346 // the quarantine or the cache because the permissions would be wrong so
347 // just unmap it.
348 unmapCallBack(Entry.MemMap);
349 break;
350 }
351
352 if (!Config::getQuarantineDisabled() && Config::getQuarantineSize()) {
353 QuarantinePos =
354 (QuarantinePos + 1) % Max(Config::getQuarantineSize(), 1u);
355 if (!Quarantine[QuarantinePos].isValid()) {
356 Quarantine[QuarantinePos] = Entry;
357 return;
358 }
359 CachedBlock PrevEntry = Quarantine[QuarantinePos];
360 Quarantine[QuarantinePos] = Entry;
361 Entry = PrevEntry;
362 }
363
364 // All excess entries are evicted from the cache. Note that when
365 // `MaxEntriesCount` is zero, cache storing shouldn't happen and it's
366 // guarded by the `DCHECK(canCache(CommitSize))` above. As a result, we
367 // won't try to pop `LRUEntries` when it's empty.
368 while (LRUEntries.size() >= atomic_load_relaxed(A: &MaxEntriesCount)) {
369 // Save MemMaps of evicted entries to perform unmap outside of lock
370 CachedBlock *Entry = LRUEntries.back();
371 EvictedCount++;
372 EvictionMemMaps.push_back(Element: Entry->MemMap);
373 remove(Entry);
374 }
375
376 insert(Entry);
377 trimResidentBytes(MaxResidentBytesLimit: atomic_load_relaxed(A: &MaxCacheResidentBytes));
378 } while (0);
379
380 for (MemMapT &EvictMemMap : EvictionMemMaps)
381 unmapCallBack(EvictMemMap);
382
383 if (Interval >= 0) {
384 // It is very likely that multiple threads trying to do a release at the
385 // same time will not actually release any extra elements. Therefore,
386 // let any other thread continue, skipping the release.
387 const u64 IntervalTime = static_cast<u64>(Interval) * 1000000;
388 if (LIKELY(Time > IntervalTime) && Mutex.tryLock()) {
389 SCUDO_SCOPED_TRACE(
390 GetSecondaryReleaseToOSTraceName(ReleaseToOS::Normal));
391
392 releaseOlderThan(ReleaseTime: Time - IntervalTime);
393 Mutex.unlock();
394 } else
395 atomic_fetch_add(A: &ReleaseToOsSkips, V: 1U, MO: memory_order_relaxed);
396 }
397 }
398
399 CachedBlock retrieve(uptr MaxAllowedFragmentedPages, uptr Size,
400 uptr Alignment, uptr HeadersSize, uptr &EntryHeaderPos)
401 EXCLUDES(Mutex) {
402 const uptr PageSize = getPageSizeCached();
403 // 10% of the requested size proved to be the optimal choice for
404 // retrieving cached blocks after testing several options.
405 constexpr u32 FragmentedBytesDivisor = 10;
406 CachedBlock Entry;
407 EntryHeaderPos = 0;
408 {
409 ScopedLock L(Mutex);
410 CallsToRetrieve++;
411 if (LRUEntries.size() == 0)
412 return {};
413 CachedBlock *RetrievedEntry = nullptr;
414 uptr MinDiff = UINTPTR_MAX;
415
416 // Since allocation sizes don't always match cached memory chunk sizes
417 // we allow some memory to be unused (called fragmented bytes). The
418 // amount of unused bytes is exactly EntryHeaderPos - CommitBase.
419 //
420 // CommitBase CommitBase + CommitSize
421 // V V
422 // +---+------------+-----------------+---+
423 // | | | | |
424 // +---+------------+-----------------+---+
425 // ^ ^ ^
426 // Guard EntryHeaderPos Guard-page-end
427 // page-begin
428 //
429 // [EntryHeaderPos, CommitBase + CommitSize) contains the user data as
430 // well as the header metadata. If EntryHeaderPos - CommitBase exceeds
431 // MaxAllowedFragmentedPages * PageSize, the cached memory chunk is
432 // not considered valid for retrieval.
433 for (CachedBlock &Entry : LRUEntries) {
434 const uptr CommitBase = Entry.CommitBase;
435 const uptr CommitSize = Entry.CommitSize;
436 const uptr AllocPos =
437 roundDown(X: CommitBase + CommitSize - Size, Boundary: Alignment);
438 const uptr HeaderPos = AllocPos - HeadersSize;
439 const uptr MaxAllowedFragmentedBytes =
440 MaxAllowedFragmentedPages * PageSize;
441 if (HeaderPos > CommitBase + CommitSize)
442 continue;
443 // TODO: Remove AllocPos > CommitBase + MaxAllowedFragmentedBytes
444 // and replace with Diff > MaxAllowedFragmentedBytes
445 if (HeaderPos < CommitBase ||
446 AllocPos > CommitBase + MaxAllowedFragmentedBytes) {
447 continue;
448 }
449
450 const uptr Diff = roundDown(X: HeaderPos, Boundary: PageSize) - CommitBase;
451
452 // Keep track of the smallest cached block
453 // that is greater than (AllocSize + HeaderSize)
454 if (Diff >= MinDiff)
455 continue;
456
457 MinDiff = Diff;
458 RetrievedEntry = &Entry;
459 EntryHeaderPos = HeaderPos;
460
461 // Immediately use a cached block if its size is close enough to the
462 // requested size
463 const uptr OptimalFitThesholdBytes =
464 (CommitBase + CommitSize - HeaderPos) / FragmentedBytesDivisor;
465 if (Diff <= OptimalFitThesholdBytes)
466 break;
467 }
468
469 if (RetrievedEntry != nullptr) {
470 Entry = *RetrievedEntry;
471 remove(Entry: RetrievedEntry);
472 SuccessfulRetrieves++;
473 }
474 }
475
476 // The difference between the retrieved memory chunk and the request
477 // size is at most MaxAllowedFragmentedPages
478 //
479 // +- MaxAllowedFragmentedPages * PageSize -+
480 // +--------------------------+-------------+
481 // | | |
482 // +--------------------------+-------------+
483 // \ Bytes to be released / ^
484 // |
485 // (may or may not be committed)
486 //
487 // The maximum number of bytes released to the OS is capped by
488 // MaxReleasedCachePages
489 //
490 // TODO : Consider making MaxReleasedCachePages configurable since
491 // the release to OS API can vary across systems.
492 if (Entry.Time != 0) {
493 const uptr FragmentedBytes =
494 roundDown(X: EntryHeaderPos, Boundary: PageSize) - Entry.CommitBase;
495 const uptr MaxUnreleasedCacheBytes = MaxUnreleasedCachePages * PageSize;
496 if (FragmentedBytes > MaxUnreleasedCacheBytes) {
497 const uptr MaxReleasedCacheBytes =
498 CachedBlock::MaxReleasedCachePages * PageSize;
499 uptr BytesToRelease =
500 roundUp(X: Min<uptr>(A: MaxReleasedCacheBytes,
501 B: FragmentedBytes - MaxUnreleasedCacheBytes),
502 Boundary: PageSize);
503 Entry.MemMap.releaseAndZeroPagesToOS(From: Entry.CommitBase, Size: BytesToRelease);
504 }
505 }
506
507 return Entry;
508 }
509
510 bool canCache(uptr Size) {
511 return atomic_load_relaxed(A: &MaxEntriesCount) != 0U &&
512 Size <= atomic_load_relaxed(A: &MaxEntrySize);
513 }
514
515 bool setOption(Option O, sptr Value) {
516 if (O == Option::ReleaseInterval) {
517 const s32 Interval = Max(
518 Min(static_cast<s32>(Value), Config::getMaxReleaseToOsIntervalMs()),
519 Config::getMinReleaseToOsIntervalMs());
520 atomic_store_relaxed(A: &ReleaseToOsIntervalMs, V: Interval);
521 if (Interval >= 0) {
522 // Always trigger a trim if the interval is not being disabled.
523 ScopedLock L(Mutex);
524 trimResidentBytes(MaxResidentBytesLimit: atomic_load_relaxed(A: &MaxCacheResidentBytes));
525 }
526 return true;
527 }
528 if (O == Option::MaxCacheEntriesCount) {
529 if (Value < 0)
530 return false;
531 atomic_store_relaxed(
532 &MaxEntriesCount,
533 Min<u32>(static_cast<u32>(Value), Config::getEntriesArraySize()));
534 return true;
535 }
536 if (O == Option::MaxCacheEntrySize) {
537 atomic_store_relaxed(A: &MaxEntrySize, V: static_cast<uptr>(Value));
538 return true;
539 }
540 if (O == Option::MaxCacheResidentBytes) {
541 if (Value < 0)
542 return false;
543 const uptr NewMaxResidentBytes = static_cast<uptr>(Value);
544 atomic_store_relaxed(A: &MaxCacheResidentBytes, V: NewMaxResidentBytes);
545 if (NewMaxResidentBytes != 0) {
546 ScopedLock L(Mutex);
547 trimResidentBytes(MaxResidentBytesLimit: NewMaxResidentBytes);
548 }
549 return true;
550 }
551 // Not supported by the Secondary Cache, but not an error either.
552 return true;
553 }
554
555 void releaseToOS([[maybe_unused]] ReleaseToOS ReleaseType) EXCLUDES(Mutex) {
556 SCUDO_SCOPED_TRACE(GetSecondaryReleaseToOSTraceName(ReleaseType));
557
558 if (ReleaseType == ReleaseToOS::ForceFast) {
559 // Never wait for the lock, always move on if there is already
560 // a release operation in progress.
561 if (Mutex.tryLock()) {
562 releaseOlderThan(UINT64_MAX);
563 Mutex.unlock();
564 }
565 } else {
566 // Since this is a request to release everything, always wait for the
567 // lock so that we guarantee all entries are released after this call.
568 ScopedLock L(Mutex);
569 releaseOlderThan(UINT64_MAX);
570 }
571 }
572
573 void disableMemoryTagging() EXCLUDES(Mutex) {
574 if (Config::getQuarantineDisabled())
575 return;
576
577 ScopedLock L(Mutex);
578 for (u32 I = 0; I != Config::getQuarantineSize(); ++I) {
579 if (Quarantine[I].isValid()) {
580 MemMapT &MemMap = Quarantine[I].MemMap;
581 unmapCallBack(MemMap);
582 Quarantine[I].invalidate();
583 }
584 }
585 QuarantinePos = -1U;
586 }
587
588 void disable() NO_THREAD_SAFETY_ANALYSIS { Mutex.lock(); }
589
590 void enable() NO_THREAD_SAFETY_ANALYSIS { Mutex.unlock(); }
591
592 uptr getMaxResidentBytesTestOnly() {
593 ScopedLock L(Mutex);
594 return MaxResidentBytes;
595 }
596
597 uptr getCurrentResidentBytesTestOnly() {
598 ScopedLock L(Mutex);
599 return CurrentResidentBytes;
600 }
601
602 void unmapTestOnly() { empty(); }
603
604 void releaseOlderThanTestOnly(u64 ReleaseTime) {
605 ScopedLock L(Mutex);
606 releaseOlderThan(ReleaseTime);
607 }
608
609private:
610 void insert(const CachedBlock &Entry) REQUIRES(Mutex) {
611 CachedBlock *AvailEntry = AvailEntries.front();
612 AvailEntries.pop_front();
613
614 *AvailEntry = Entry;
615 LRUEntries.push_front(X: AvailEntry);
616 if (OldestPresentEntry == nullptr && AvailEntry->Time != 0)
617 OldestPresentEntry = AvailEntry;
618 if (AvailEntry->Time != 0) {
619 CurrentResidentBytes += Entry.CommitSize;
620 if (CurrentResidentBytes > MaxResidentBytes)
621 MaxResidentBytes = CurrentResidentBytes;
622 }
623 }
624
625 void remove(CachedBlock *Entry) REQUIRES(Mutex) {
626 DCHECK(Entry->isValid());
627 if (OldestPresentEntry == Entry) {
628 OldestPresentEntry = LRUEntries.getPrev(X: Entry);
629 DCHECK(OldestPresentEntry == nullptr || OldestPresentEntry->Time != 0);
630 }
631 LRUEntries.remove(X: Entry);
632 if (Entry->Time != 0)
633 CurrentResidentBytes -= Entry->CommitSize;
634 Entry->invalidate();
635 AvailEntries.push_front(X: Entry);
636 }
637
638 ALWAYS_INLINE void trimResidentBytes(uptr MaxResidentBytesLimit)
639 REQUIRES(Mutex) {
640 if (MaxResidentBytesLimit == 0 ||
641 atomic_load_relaxed(A: &ReleaseToOsIntervalMs) < 0)
642 return;
643 while (CurrentResidentBytes > MaxResidentBytesLimit &&
644 OldestPresentEntry != nullptr) {
645 CachedBlock *Entry = OldestPresentEntry;
646 OldestPresentEntry = LRUEntries.getPrev(X: Entry);
647 Entry->MemMap.releaseAndZeroPagesToOS(From: Entry->CommitBase,
648 Size: Entry->CommitSize);
649 CurrentResidentBytes -= Entry->CommitSize;
650 Entry->Time = 0;
651 }
652 }
653
654 void empty() {
655 MemMapT MapInfo[Config::getEntriesArraySize()];
656 uptr N = 0;
657 {
658 ScopedLock L(Mutex);
659
660 for (CachedBlock &Entry : LRUEntries)
661 MapInfo[N++] = Entry.MemMap;
662 LRUEntries.clear();
663 OldestPresentEntry = nullptr;
664 CurrentResidentBytes = 0;
665 }
666 for (uptr I = 0; I < N; I++) {
667 MemMapT &MemMap = MapInfo[I];
668 unmapCallBack(MemMap);
669 }
670 }
671
672 void releaseOlderThan(u64 ReleaseTime) REQUIRES(Mutex) {
673 SCUDO_SCOPED_TRACE(GetSecondaryReleaseOlderThanTraceName());
674
675 if (!Config::getQuarantineDisabled()) {
676 for (uptr I = 0; I < Config::getQuarantineSize(); I++) {
677 auto &Entry = Quarantine[I];
678 if (!Entry.isValid() || Entry.Time == 0 || Entry.Time > ReleaseTime)
679 continue;
680 Entry.MemMap.releaseAndZeroPagesToOS(Entry.CommitBase,
681 Entry.CommitSize);
682 Entry.Time = 0;
683 }
684 }
685
686 for (CachedBlock *Entry = OldestPresentEntry; Entry != nullptr;
687 Entry = LRUEntries.getPrev(X: Entry)) {
688 DCHECK(Entry->isValid());
689 DCHECK(Entry->Time != 0);
690
691 if (Entry->Time > ReleaseTime) {
692 // All entries are newer than this, so no need to keep scanning.
693 OldestPresentEntry = Entry;
694 return;
695 }
696
697 Entry->MemMap.releaseAndZeroPagesToOS(From: Entry->CommitBase,
698 Size: Entry->CommitSize);
699 CurrentResidentBytes -= Entry->CommitSize;
700 Entry->Time = 0;
701 }
702 OldestPresentEntry = nullptr;
703 }
704
705 HybridMutex Mutex;
706 u32 QuarantinePos GUARDED_BY(Mutex) = 0;
707 atomic_u32 MaxEntriesCount = {};
708 atomic_uptr MaxEntrySize = {};
709 atomic_uptr MaxCacheResidentBytes = {};
710 atomic_s32 ReleaseToOsIntervalMs = {};
711 u32 CallsToRetrieve GUARDED_BY(Mutex) = 0;
712 u32 SuccessfulRetrieves GUARDED_BY(Mutex) = 0;
713 u32 EvictedCount GUARDED_BY(Mutex) = 0;
714 uptr CurrentResidentBytes GUARDED_BY(Mutex) = 0;
715 uptr MaxResidentBytes GUARDED_BY(Mutex) = 0;
716 atomic_uptr ReleaseToOsSkips = {};
717
718 CachedBlock Entries[Config::getEntriesArraySize()] GUARDED_BY(Mutex) = {};
719 NonZeroLengthArray<CachedBlock, Config::getQuarantineSize()>
720 Quarantine GUARDED_BY(Mutex) = {};
721
722 // The oldest entry in the LRUEntries that has Time non-zero.
723 CachedBlock *OldestPresentEntry GUARDED_BY(Mutex) = nullptr;
724 // Cached blocks stored in LRU order
725 DoublyLinkedList<CachedBlock> LRUEntries GUARDED_BY(Mutex);
726 // The unused Entries
727 SinglyLinkedList<CachedBlock> AvailEntries GUARDED_BY(Mutex);
728};
729
730template <typename Config> class MapAllocator {
731public:
732 void init(GlobalStats *S,
733 s32 ReleaseToOsInterval = -1) NO_THREAD_SAFETY_ANALYSIS {
734 DCHECK_EQ(AllocatedBytes, 0U);
735 DCHECK_EQ(FreedBytes, 0U);
736 Cache.init(ReleaseToOsInterval);
737 Stats.init();
738 if (LIKELY(S))
739 S->link(S: &Stats);
740 }
741
742 void *allocate(const Options &Options, uptr Size, uptr AlignmentHint = 0,
743 uptr *BlockEnd = nullptr,
744 FillContentsMode FillContents = NoFill);
745
746 void deallocate(const Options &Options, void *Ptr);
747
748 void *tryAllocateFromCache(const Options &Options, uptr Size, uptr Alignment,
749 uptr *BlockEndPtr, FillContentsMode FillContents);
750
751 static uptr getBlockEnd(void *Ptr) {
752 auto *B = LargeBlock::getHeader<Config>(Ptr);
753 return B->CommitBase + B->CommitSize;
754 }
755
756 static uptr getBlockSize(void *Ptr) {
757 return getBlockEnd(Ptr) - reinterpret_cast<uptr>(Ptr);
758 }
759
760 static uptr getGuardPageSize() {
761 if (Config::getEnableGuardPages())
762 return getPageSizeCached();
763 return 0U;
764 }
765
766 static constexpr uptr getHeadersSize() {
767 return Chunk::getHeaderSize() + LargeBlock::getHeaderSize();
768 }
769
770 void disable() NO_THREAD_SAFETY_ANALYSIS {
771 Mutex.lock();
772 Cache.disable();
773 }
774
775 void enable() NO_THREAD_SAFETY_ANALYSIS {
776 Cache.enable();
777 Mutex.unlock();
778 }
779
780 template <typename F> void iterateOverBlocks(F Callback) const {
781 Mutex.assertHeld();
782
783 for (const auto &H : InUseBlocks) {
784 uptr Ptr = reinterpret_cast<uptr>(&H) + LargeBlock::getHeaderSize();
785 if (allocatorSupportsMemoryTagging<Config>())
786 Ptr = untagPointer(Ptr);
787 Callback(Ptr);
788 }
789 }
790
791 bool canCache(uptr Size) { return Cache.canCache(Size); }
792
793 bool setOption(Option O, sptr Value) { return Cache.setOption(O, Value); }
794
795 void releaseToOS(ReleaseToOS ReleaseType) { Cache.releaseToOS(ReleaseType); }
796
797 void disableMemoryTagging() { Cache.disableMemoryTagging(); }
798
799 void unmapTestOnly() { Cache.unmapTestOnly(); }
800
801 uptr getMaxResidentBytesTestOnly() {
802 return Cache.getMaxResidentBytesTestOnly();
803 }
804
805 uptr getCurrentResidentBytesTestOnly() {
806 return Cache.getCurrentResidentBytesTestOnly();
807 }
808
809 void getStats(ScopedString *Str);
810
811private:
812 typename Config::template CacheT<typename Config::CacheConfig> Cache;
813
814 mutable HybridMutex Mutex;
815 DoublyLinkedList<LargeBlock::Header> InUseBlocks GUARDED_BY(Mutex);
816 uptr AllocatedBytes GUARDED_BY(Mutex) = 0;
817 uptr FreedBytes GUARDED_BY(Mutex) = 0;
818 uptr FragmentedBytes GUARDED_BY(Mutex) = 0;
819 uptr LargestSize GUARDED_BY(Mutex) = 0;
820 u32 UncacheableUnmaps GUARDED_BY(Mutex) = 0;
821 u32 NumberOfAllocs GUARDED_BY(Mutex) = 0;
822 u32 NumberOfFrees GUARDED_BY(Mutex) = 0;
823 LocalStats Stats GUARDED_BY(Mutex);
824};
825
826template <typename Config>
827void *
828MapAllocator<Config>::tryAllocateFromCache(const Options &Options, uptr Size,
829 uptr Alignment, uptr *BlockEndPtr,
830 FillContentsMode FillContents) {
831 CachedBlock Entry;
832 uptr EntryHeaderPos;
833 uptr MaxAllowedFragmentedPages = MaxUnreleasedCachePages;
834
835 if (LIKELY(!useMemoryTagging<Config>(Options))) {
836 MaxAllowedFragmentedPages += CachedBlock::MaxReleasedCachePages;
837 } else {
838 // TODO: Enable MaxReleasedCachePages may result in pages for an entry being
839 // partially released and it erases the tag of those pages as well. To
840 // support this feature for MTE, we need to tag those pages again.
841 DCHECK_EQ(MaxAllowedFragmentedPages, MaxUnreleasedCachePages);
842 }
843
844 Entry = Cache.retrieve(MaxAllowedFragmentedPages, Size, Alignment,
845 getHeadersSize(), EntryHeaderPos);
846 if (!Entry.isValid())
847 return nullptr;
848
849 LargeBlock::Header *H = reinterpret_cast<LargeBlock::Header *>(
850 LargeBlock::addHeaderTag<Config>(EntryHeaderPos));
851 bool Zeroed = Entry.Time == 0;
852
853 if (UNLIKELY(Entry.Flags & CachedBlock::NoAccess)) {
854 // NOTE: Flags set to 0 actually restores read-write.
855 Entry.MemMap.setMemoryPermission(Addr: Entry.CommitBase, Size: Entry.CommitSize,
856 /*Flags=*/Flags: 0);
857 }
858
859 if (useMemoryTagging<Config>(Options)) {
860 const uptr PageSize = getPageSizeCached();
861 const uptr OldAllocPos =
862 untagPointer(Ptr: Entry.BlockBegin) + Chunk::getHeaderSize();
863 const uptr NewAllocPos = untagPointer(Ptr: roundUp(X: EntryHeaderPos, Boundary: Alignment));
864 DCHECK_GE(Alignment, PageSize);
865
866 const uptr OldHeaderPage = OldAllocPos - PageSize;
867 const uptr NewHeaderPage = NewAllocPos - PageSize;
868
869 // Enabling or disabling memory tagging at runtime is unsupported.
870 // If MTE is enabled now, the cached entry was also allocated with MTE
871 // enabled, guaranteeing that both OldAllocPos and NewAllocPos are
872 // page-aligned.
873 CHECK(OldAllocPos % PageSize == 0U);
874 DCHECK(NewAllocPos % PageSize == 0U);
875
876 if (NewAllocPos != OldAllocPos) {
877 // The shift distance must be a multiple of PageSize
878 DCHECK_EQ((NewAllocPos > OldAllocPos ? NewAllocPos - OldAllocPos
879 : OldAllocPos - NewAllocPos) %
880 PageSize,
881 0U);
882
883 const uptr MappingFlags = MAP_RESIZABLE | MAP_ALLOWNOMEM;
884
885 if (!Entry.MemMap.remap(Addr: NewHeaderPage, Size: PageSize, Name: "scudo:secondary",
886 MAP_MEMTAG | MappingFlags)) {
887 unmap(MemMap&: Entry.MemMap);
888 return nullptr;
889 }
890 // Since PROT_MTE is sticky, setting memory permissions to MAP_NOACCESS
891 // (PROT_NONE) when caching the block does not clear the PROT_MTE flag
892 // from the kernel VMA. When we make the block RW again, the old header
893 // page still has MTE enabled. If the allocation shifted, we must
894 // explicitly remap the old header page without MAP_MEMTAG to disable MTE,
895 // as this page may now be part of the user payload or padding.
896 if (!Entry.MemMap.remap(Addr: OldHeaderPage, Size: PageSize, Name: "scudo:secondary",
897 Flags: MappingFlags)) {
898 unmap(MemMap&: Entry.MemMap);
899 return nullptr;
900 }
901 }
902
903 uptr NewBlockBegin = reinterpret_cast<uptr>(H + 1);
904 storeTags(Begin: reinterpret_cast<uptr>(H), End: NewBlockBegin);
905 }
906
907 H->CommitBase = Entry.CommitBase;
908 H->CommitSize = Entry.CommitSize;
909 H->MemMap = Entry.MemMap;
910
911 const uptr BlockEnd = H->CommitBase + H->CommitSize;
912 if (BlockEndPtr)
913 *BlockEndPtr = BlockEnd;
914 uptr HInt = reinterpret_cast<uptr>(H);
915 if (allocatorSupportsMemoryTagging<Config>())
916 HInt = untagPointer(Ptr: HInt);
917 const uptr PtrInt = HInt + LargeBlock::getHeaderSize();
918 void *Ptr = reinterpret_cast<void *>(PtrInt);
919 if (FillContents && !Zeroed)
920 memset(s: Ptr, c: FillContents == ZeroFill ? 0 : PatternFillByte,
921 n: BlockEnd - PtrInt);
922 {
923 ScopedLock L(Mutex);
924 InUseBlocks.push_back(X: H);
925 AllocatedBytes += H->CommitSize;
926 FragmentedBytes += H->MemMap.getCapacity() - H->CommitSize;
927 NumberOfAllocs++;
928 Stats.add(I: StatAllocated, V: H->CommitSize);
929 Stats.add(I: StatMapped, V: H->MemMap.getCapacity());
930 }
931 return Ptr;
932}
933// As with the Primary, the size passed to this function includes any desired
934// alignment, so that the frontend can align the user allocation. The hint
935// parameter allows us to unmap spurious memory when dealing with larger
936// (greater than a page) alignments on 32-bit platforms.
937// Due to the sparsity of address space available on those platforms, requesting
938// an allocation from the Secondary with a large alignment would end up wasting
939// VA space (even though we are not committing the whole thing), hence the need
940// to trim off some of the reserved space.
941// For allocations requested with an alignment greater than or equal to a page,
942// the committed memory will amount to something close to Size - AlignmentHint
943// (pending rounding and headers).
944template <typename Config>
945void *MapAllocator<Config>::allocate(const Options &Options, uptr Size,
946 uptr Alignment, uptr *BlockEndPtr,
947 FillContentsMode FillContents) {
948 if (Options.get(Opt: OptionBit::AddLargeAllocationSlack))
949 Size += 1UL << SCUDO_MIN_ALIGNMENT_LOG;
950 Alignment = Max(A: Alignment, B: uptr(1U) << SCUDO_MIN_ALIGNMENT_LOG);
951 const uptr PageSize = getPageSizeCached();
952
953 if (useMemoryTagging<Config>(Options))
954 Alignment = Max(A: Alignment, B: PageSize);
955
956 // Note that cached blocks may have aligned address already. Thus we simply
957 // pass the required size (`Size` + `getHeadersSize()`) to do cache look up.
958 const uptr MinNeededSizeForCache = roundUp(X: Size + getHeadersSize(), Boundary: PageSize);
959
960 if (Alignment <= PageSize && Cache.canCache(MinNeededSizeForCache)) {
961 void *Ptr = tryAllocateFromCache(Options, Size, Alignment, BlockEndPtr,
962 FillContents);
963 if (Ptr != nullptr)
964 return Ptr;
965 }
966
967 uptr RoundedSize =
968 roundUp(X: roundUp(X: Size, Boundary: Alignment) + getHeadersSize(), Boundary: PageSize);
969 if (UNLIKELY(Alignment > PageSize))
970 RoundedSize += Alignment - PageSize;
971
972 ReservedMemoryT ReservedMemory;
973 const uptr MapSize = RoundedSize + 2 * getGuardPageSize();
974 if (UNLIKELY(!ReservedMemory.create(/*Addr=*/0U, MapSize, nullptr,
975 MAP_ALLOWNOMEM))) {
976 return nullptr;
977 }
978
979 // Take the entire ownership of reserved region.
980 MemMapT MemMap = ReservedMemory.dispatch(Addr: ReservedMemory.getBase(),
981 Size: ReservedMemory.getCapacity());
982 uptr MapBase = MemMap.getBase();
983 uptr CommitBase = MapBase + getGuardPageSize();
984 uptr MapEnd = MapBase + MapSize;
985
986 // In the unlikely event of alignments larger than a page, adjust the amount
987 // of memory we want to commit, and trim the extra memory.
988 if (UNLIKELY(Alignment >= PageSize)) {
989 // For alignments greater than or equal to a page, the user pointer (eg:
990 // the pointer that is returned by the C or C++ allocation APIs) ends up
991 // on a page boundary , and our headers will live in the preceding page.
992 CommitBase =
993 roundUp(X: MapBase + getGuardPageSize() + 1, Boundary: Alignment) - PageSize;
994 // We only trim the extra memory on 32-bit platforms: 64-bit platforms
995 // are less constrained memory wise, and that saves us two syscalls.
996 if (SCUDO_WORDSIZE == 32U) {
997 const uptr NewMapBase = CommitBase - getGuardPageSize();
998 DCHECK_GE(NewMapBase, MapBase);
999 if (NewMapBase != MapBase) {
1000 MemMap.unmap(Addr: MapBase, Size: NewMapBase - MapBase);
1001 MapBase = NewMapBase;
1002 }
1003 // CommitBase is past the first guard page, but this computation needs
1004 // to include a page where the header lives.
1005 const uptr NewMapEnd =
1006 CommitBase + PageSize + roundUp(X: Size, Boundary: PageSize) + getGuardPageSize();
1007 DCHECK_LE(NewMapEnd, MapEnd);
1008 if (NewMapEnd != MapEnd) {
1009 MemMap.unmap(Addr: NewMapEnd, Size: MapEnd - NewMapEnd);
1010 MapEnd = NewMapEnd;
1011 }
1012 }
1013 }
1014
1015 const uptr CommitSize = MapEnd - getGuardPageSize() - CommitBase;
1016 const uptr AllocPos = roundDown(X: CommitBase + CommitSize - Size, Boundary: Alignment);
1017 if (!mapSecondary<Config>(Options, CommitBase, CommitSize, AllocPos, 0,
1018 MemMap)) {
1019 MemMap.unmap();
1020 return nullptr;
1021 }
1022 const uptr HeaderPos = AllocPos - getHeadersSize();
1023 // Make sure that the header is not in the guard page or before the base.
1024 DCHECK_GE(HeaderPos, MapBase + getGuardPageSize());
1025 LargeBlock::Header *H = reinterpret_cast<LargeBlock::Header *>(
1026 LargeBlock::addHeaderTag<Config>(HeaderPos));
1027 if (useMemoryTagging<Config>(Options))
1028 storeTags(Begin: reinterpret_cast<uptr>(H), End: reinterpret_cast<uptr>(H + 1));
1029 H->CommitBase = CommitBase;
1030 H->CommitSize = CommitSize;
1031 H->MemMap = MemMap;
1032 if (BlockEndPtr)
1033 *BlockEndPtr = CommitBase + CommitSize;
1034 {
1035 ScopedLock L(Mutex);
1036 InUseBlocks.push_back(X: H);
1037 AllocatedBytes += CommitSize;
1038 FragmentedBytes += H->MemMap.getCapacity() - CommitSize;
1039 if (LargestSize < CommitSize)
1040 LargestSize = CommitSize;
1041 NumberOfAllocs++;
1042 Stats.add(I: StatAllocated, V: CommitSize);
1043 Stats.add(I: StatMapped, V: H->MemMap.getCapacity());
1044 }
1045 return reinterpret_cast<void *>(HeaderPos + LargeBlock::getHeaderSize());
1046}
1047
1048template <typename Config>
1049void MapAllocator<Config>::deallocate(const Options &Options, void *Ptr)
1050 EXCLUDES(Mutex) {
1051 LargeBlock::Header *H = LargeBlock::getHeader<Config>(Ptr);
1052 const uptr CommitSize = H->CommitSize;
1053 {
1054 ScopedLock L(Mutex);
1055 InUseBlocks.remove(X: H);
1056 FreedBytes += CommitSize;
1057 FragmentedBytes -= H->MemMap.getCapacity() - CommitSize;
1058 NumberOfFrees++;
1059 Stats.sub(I: StatAllocated, V: CommitSize);
1060 Stats.sub(I: StatMapped, V: H->MemMap.getCapacity());
1061 }
1062
1063 if (Cache.canCache(H->CommitSize)) {
1064 Cache.store(Options, H->CommitBase, H->CommitSize,
1065 reinterpret_cast<uptr>(H + 1), H->MemMap);
1066 } else {
1067 // Note that the `H->MemMap` is stored on the pages managed by itself. Take
1068 // over the ownership before unmap() so that any operation along with
1069 // unmap() won't touch inaccessible pages.
1070 MemMapT MemMap = H->MemMap;
1071 unmap(MemMap);
1072 ScopedLock L(Mutex);
1073 UncacheableUnmaps++;
1074 }
1075}
1076
1077template <typename Config>
1078void MapAllocator<Config>::getStats(ScopedString *Str) EXCLUDES(Mutex) {
1079 ScopedLock L(Mutex);
1080 Str->append(
1081 Format: "Stats: MapAllocator: allocated %u times (%zuK), freed %u times (%zuK), "
1082 "remains %u (%zuK) max %zuM, Fragmented %zuK, Uncacheable unmaps: %u\n",
1083 NumberOfAllocs, AllocatedBytes >> 10, NumberOfFrees, FreedBytes >> 10,
1084 NumberOfAllocs - NumberOfFrees, (AllocatedBytes - FreedBytes) >> 10,
1085 LargestSize >> 20, FragmentedBytes >> 10, UncacheableUnmaps);
1086 Cache.getStats(Str);
1087}
1088
1089} // namespace scudo
1090
1091#endif // SCUDO_SECONDARY_H_
1092