1//===- AtomicExpandPass.cpp - Expand atomic instructions ------------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file contains a pass (at IR level) to replace atomic instructions with
10// __atomic_* library calls, or target specific instruction which implement the
11// same semantics in a way which better fits the target backend. This can
12// include the use of (intrinsic-based) load-linked/store-conditional loops,
13// AtomicCmpXchg, or type coercions.
14//
15//===----------------------------------------------------------------------===//
16
17#include "llvm/ADT/ArrayRef.h"
18#include "llvm/ADT/STLFunctionalExtras.h"
19#include "llvm/ADT/SmallString.h"
20#include "llvm/ADT/SmallVector.h"
21#include "llvm/Analysis/InstSimplifyFolder.h"
22#include "llvm/Analysis/OptimizationRemarkEmitter.h"
23#include "llvm/CodeGen/AtomicExpand.h"
24#include "llvm/CodeGen/TargetLowering.h"
25#include "llvm/CodeGen/TargetPassConfig.h"
26#include "llvm/CodeGen/TargetSubtargetInfo.h"
27#include "llvm/CodeGen/ValueTypes.h"
28#include "llvm/IR/Attributes.h"
29#include "llvm/IR/BasicBlock.h"
30#include "llvm/IR/Constant.h"
31#include "llvm/IR/Constants.h"
32#include "llvm/IR/DataLayout.h"
33#include "llvm/IR/DerivedTypes.h"
34#include "llvm/IR/Function.h"
35#include "llvm/IR/IRBuilder.h"
36#include "llvm/IR/Instruction.h"
37#include "llvm/IR/Instructions.h"
38#include "llvm/IR/MDBuilder.h"
39#include "llvm/IR/MemoryModelRelaxationAnnotations.h"
40#include "llvm/IR/Module.h"
41#include "llvm/IR/ProfDataUtils.h"
42#include "llvm/IR/Type.h"
43#include "llvm/IR/User.h"
44#include "llvm/IR/Value.h"
45#include "llvm/InitializePasses.h"
46#include "llvm/Pass.h"
47#include "llvm/Support/AtomicOrdering.h"
48#include "llvm/Support/Casting.h"
49#include "llvm/Support/Debug.h"
50#include "llvm/Support/ErrorHandling.h"
51#include "llvm/Support/raw_ostream.h"
52#include "llvm/Target/TargetMachine.h"
53#include "llvm/Transforms/Utils/LowerAtomic.h"
54#include <cassert>
55#include <cstdint>
56#include <iterator>
57
58using namespace llvm;
59
60#define DEBUG_TYPE "atomic-expand"
61
62namespace {
63
64class AtomicExpandImpl {
65 const TargetLowering *TLI = nullptr;
66 const LibcallLoweringInfo *LibcallLowering = nullptr;
67 const DataLayout *DL = nullptr;
68 bool SingleThreaded = false;
69
70private:
71 /// Callback type for emitting a cmpxchg instruction during RMW expansion.
72 /// Parameters: (Builder, Addr, Loaded, NewVal, AddrAlign, MemOpOrder,
73 /// SSID, IsVolatile, /* OUT */ Success, /* OUT */ NewLoaded,
74 /// MetadataSrc)
75 using CreateCmpXchgInstFun = function_ref<void(
76 IRBuilderBase &, Value *, Value *, Value *, Align, AtomicOrdering,
77 SyncScope::ID, bool, Value *&, Value *&, Instruction *)>;
78
79 void handleFailure(Instruction &FailedInst, const Twine &Msg,
80 Instruction *DiagnosticInst = nullptr) const {
81 LLVMContext &Ctx = FailedInst.getContext();
82
83 // TODO: Do not use generic error type.
84 Ctx.emitError(I: DiagnosticInst ? DiagnosticInst : &FailedInst, ErrorStr: Msg);
85
86 if (!FailedInst.getType()->isVoidTy())
87 FailedInst.replaceAllUsesWith(V: PoisonValue::get(T: FailedInst.getType()));
88 FailedInst.eraseFromParent();
89 }
90
91 template <typename Inst>
92 void handleUnsupportedAtomicSize(Inst *I, const Twine &AtomicOpName,
93 Instruction *DiagnosticInst = nullptr) const;
94
95 bool bracketInstWithFences(Instruction *I, AtomicOrdering Order);
96 bool tryInsertTrailingSeqCstFence(Instruction *AtomicI);
97 template <typename AtomicInst>
98 bool tryInsertFencesForAtomic(AtomicInst *AtomicI, bool OrderingRequiresFence,
99 AtomicOrdering NewOrdering);
100 IntegerType *getCorrespondingIntegerType(Type *T, const DataLayout &DL);
101 LoadInst *convertAtomicLoadToIntegerType(LoadInst *LI);
102 bool tryExpandAtomicLoad(LoadInst *LI);
103 bool expandAtomicLoadToLL(LoadInst *LI);
104 bool expandAtomicLoadToCmpXchg(LoadInst *LI);
105 StoreInst *convertAtomicStoreToIntegerType(StoreInst *SI);
106 bool tryExpandAtomicStore(StoreInst *SI);
107 void expandAtomicStoreToXChg(StoreInst *SI);
108 bool tryExpandAtomicRMW(AtomicRMWInst *AI);
109 void expandAtomicSubToAdd(AtomicRMWInst *AI);
110 AtomicRMWInst *convertAtomicXchgToIntegerType(AtomicRMWInst *RMWI);
111 Value *
112 insertRMWLLSCLoop(IRBuilderBase &Builder, Type *ResultTy, Value *Addr,
113 Align AddrAlign, AtomicOrdering MemOpOrder,
114 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp);
115 void expandAtomicOpToLLSC(
116 Instruction *I, Type *ResultTy, Value *Addr, Align AddrAlign,
117 AtomicOrdering MemOpOrder,
118 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp);
119 void expandPartwordAtomicRMW(
120 AtomicRMWInst *I, TargetLoweringBase::AtomicExpansionKind ExpansionKind);
121 AtomicRMWInst *widenPartwordAtomicRMW(AtomicRMWInst *AI);
122 bool expandPartwordCmpXchg(AtomicCmpXchgInst *I);
123 void expandAtomicRMWToMaskedIntrinsic(AtomicRMWInst *AI);
124 void expandAtomicCmpXchgToMaskedIntrinsic(AtomicCmpXchgInst *CI);
125
126 AtomicCmpXchgInst *convertCmpXchgToIntegerType(AtomicCmpXchgInst *CI);
127 Value *insertRMWCmpXchgLoop(
128 IRBuilderBase &Builder, Type *ResultType, Value *Addr, Align AddrAlign,
129 AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile,
130 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp,
131 CreateCmpXchgInstFun CreateCmpXchg, Instruction *MetadataSrc);
132 bool tryExpandAtomicCmpXchg(AtomicCmpXchgInst *CI);
133
134 bool expandAtomicCmpXchg(AtomicCmpXchgInst *CI);
135 bool isIdempotentRMW(AtomicRMWInst *RMWI);
136 bool simplifyIdempotentRMW(AtomicRMWInst *RMWI);
137
138 bool expandAtomicOpToLibcall(Instruction *I, unsigned Size, Align Alignment,
139 Value *PointerOperand, Value *ValueOperand,
140 Value *CASExpected, AtomicOrdering Ordering,
141 AtomicOrdering Ordering2,
142 ArrayRef<RTLIB::Libcall> Libcalls);
143 void expandAtomicLoadToLibcall(LoadInst *LI);
144 void expandAtomicStoreToLibcall(StoreInst *LI);
145 void expandAtomicRMWToLibcall(AtomicRMWInst *I);
146 void expandAtomicCASToLibcall(AtomicCmpXchgInst *I,
147 const Twine &AtomicOpName = "cmpxchg",
148 Instruction *DiagnosticInst = nullptr);
149
150 bool expandAtomicRMWToCmpXchg(AtomicRMWInst *AI,
151 CreateCmpXchgInstFun CreateCmpXchg);
152
153 bool lowerToNonAtomic(Instruction *I);
154 bool processAtomicInstr(Instruction *I);
155
156public:
157 bool run(Function &F, const ModuleLibcallLoweringInfo &LibcallResult,
158 const TargetMachine *TM);
159};
160
161class AtomicExpandLegacy : public FunctionPass {
162public:
163 static char ID; // Pass identification, replacement for typeid
164
165 AtomicExpandLegacy() : FunctionPass(ID) {}
166
167 void getAnalysisUsage(AnalysisUsage &AU) const override {
168 AU.addRequired<LibcallLoweringInfoWrapper>();
169 FunctionPass::getAnalysisUsage(AU);
170 }
171
172 bool runOnFunction(Function &F) override;
173};
174
175// IRBuilder to be used for replacement atomic instructions.
176struct ReplacementIRBuilder
177 : IRBuilder<InstSimplifyFolder, IRBuilderCallbackInserter> {
178 MDNode *MMRAMD = nullptr;
179 MDNode *PCSectionsMD = nullptr;
180
181 // Preserves the DebugLoc from I, and preserves still valid metadata.
182 // Enable StrictFP builder mode when appropriate.
183 explicit ReplacementIRBuilder(Instruction *I, const DataLayout &DL)
184 : IRBuilder(
185 I->getIterator(), InstSimplifyFolder(DL),
186 IRBuilderCallbackInserter([this](Instruction *I) { addMD(I); })) {
187 if (BB->getParent()->getAttributes().hasFnAttr(Kind: Attribute::StrictFP))
188 this->setIsFPConstrained(true);
189
190 MMRAMD = I->getMetadata(KindID: LLVMContext::MD_mmra);
191 PCSectionsMD = I->getMetadata(KindID: LLVMContext::MD_pcsections);
192 }
193
194 void addMD(Instruction *I) {
195 if (canInstructionHaveMMRAs(I: *I))
196 I->setMetadata(KindID: LLVMContext::MD_mmra, Node: MMRAMD);
197 I->setMetadata(KindID: LLVMContext::MD_pcsections, Node: PCSectionsMD);
198 }
199};
200
201} // end anonymous namespace
202
203char AtomicExpandLegacy::ID = 0;
204
205char &llvm::AtomicExpandID = AtomicExpandLegacy::ID;
206
207INITIALIZE_PASS_BEGIN(AtomicExpandLegacy, DEBUG_TYPE,
208 "Expand Atomic instructions", false, false)
209INITIALIZE_PASS_DEPENDENCY(LibcallLoweringInfoWrapper)
210INITIALIZE_PASS_DEPENDENCY(TargetPassConfig)
211INITIALIZE_PASS_END(AtomicExpandLegacy, DEBUG_TYPE,
212 "Expand Atomic instructions", false, false)
213
214// Helper functions to retrieve the size of atomic instructions.
215static unsigned getAtomicOpSize(LoadInst *LI) {
216 const DataLayout &DL = LI->getDataLayout();
217 return DL.getTypeStoreSize(Ty: LI->getType());
218}
219
220static unsigned getAtomicOpSize(StoreInst *SI) {
221 const DataLayout &DL = SI->getDataLayout();
222 return DL.getTypeStoreSize(Ty: SI->getValueOperand()->getType());
223}
224
225static unsigned getAtomicOpSize(AtomicRMWInst *RMWI) {
226 const DataLayout &DL = RMWI->getDataLayout();
227 return DL.getTypeStoreSize(Ty: RMWI->getValOperand()->getType());
228}
229
230static unsigned getAtomicOpSize(AtomicCmpXchgInst *CASI) {
231 const DataLayout &DL = CASI->getDataLayout();
232 return DL.getTypeStoreSize(Ty: CASI->getCompareOperand()->getType());
233}
234
235/// Copy metadata that's safe to preserve when widening atomics.
236static void copyMetadataForAtomic(Instruction &Dest,
237 const Instruction &Source) {
238 SmallVector<std::pair<unsigned, MDNode *>, 8> MD;
239 Source.getAllMetadata(MDs&: MD);
240 LLVMContext &Ctx = Dest.getContext();
241 MDBuilder MDB(Ctx);
242
243 for (auto [ID, N] : MD) {
244 switch (ID) {
245 case LLVMContext::MD_dbg:
246 case LLVMContext::MD_tbaa:
247 case LLVMContext::MD_tbaa_struct:
248 case LLVMContext::MD_alias_scope:
249 case LLVMContext::MD_mem_cache_hint:
250 case LLVMContext::MD_noalias:
251 case LLVMContext::MD_noalias_addrspace:
252 case LLVMContext::MD_access_group:
253 case LLVMContext::MD_mmra:
254 Dest.setMetadata(KindID: ID, Node: N);
255 break;
256 default:
257 if (ID == Ctx.getMDKindID(Name: "amdgpu.no.remote.memory"))
258 Dest.setMetadata(KindID: ID, Node: N);
259 else if (ID == Ctx.getMDKindID(Name: "amdgpu.no.fine.grained.memory"))
260 Dest.setMetadata(KindID: ID, Node: N);
261
262 // Losing atomic.ignore.denormal.mode, but it doesn't matter for current
263 // uses.
264 break;
265 }
266 }
267}
268
269template <typename Inst>
270static bool atomicSizeSupported(const TargetLowering *TLI, Inst *I) {
271 unsigned Size = getAtomicOpSize(I);
272 Align Alignment = I->getAlign();
273 unsigned MaxSize = TLI->getMaxAtomicSizeInBitsSupported() / 8;
274 return Alignment >= Size && Size <= MaxSize;
275}
276
277template <typename Inst>
278static void writeUnsupportedAtomicSizeReason(const TargetLowering *TLI, Inst *I,
279 raw_ostream &OS) {
280 unsigned Size = getAtomicOpSize(I);
281 Align Alignment = I->getAlign();
282 bool NeedSeparator = false;
283
284 if (Alignment < Size) {
285 OS << "instruction alignment " << Alignment.value()
286 << " is smaller than the required " << Size
287 << "-byte alignment for this atomic operation";
288 NeedSeparator = true;
289 }
290
291 unsigned MaxSize = TLI->getMaxAtomicSizeInBitsSupported() / 8;
292 if (Size > MaxSize) {
293 if (NeedSeparator)
294 OS << "; ";
295 OS << "target supports atomics up to " << MaxSize
296 << " bytes, but this atomic accesses " << Size << " bytes";
297 }
298}
299
300template <typename Inst>
301void AtomicExpandImpl::handleUnsupportedAtomicSize(
302 Inst *I, const Twine &AtomicOpName, Instruction *DiagnosticInst) const {
303 assert(!atomicSizeSupported(TLI, I) && "expected unsupported atomic size");
304 SmallString<128> FailureReason;
305 raw_svector_ostream OS(FailureReason);
306 writeUnsupportedAtomicSizeReason(TLI, I, OS);
307 handleFailure(FailedInst&: *I, Msg: Twine("unsupported ") + AtomicOpName + ": " + FailureReason,
308 DiagnosticInst);
309}
310
311bool AtomicExpandImpl::tryInsertTrailingSeqCstFence(Instruction *AtomicI) {
312 if (!TLI->shouldInsertTrailingSeqCstFenceForAtomicStore(I: AtomicI))
313 return false;
314
315 IRBuilder Builder(AtomicI);
316 if (auto *TrailingFence = TLI->emitTrailingFence(
317 Builder, Inst: AtomicI, Ord: AtomicOrdering::SequentiallyConsistent)) {
318 TrailingFence->moveAfter(MovePos: AtomicI);
319 return true;
320 }
321 return false;
322}
323
324template <typename AtomicInst>
325bool AtomicExpandImpl::tryInsertFencesForAtomic(AtomicInst *AtomicI,
326 bool OrderingRequiresFence,
327 AtomicOrdering NewOrdering) {
328 bool ShouldInsertFences = TLI->shouldInsertFencesForAtomic(I: AtomicI);
329 if (OrderingRequiresFence && ShouldInsertFences) {
330 AtomicOrdering FenceOrdering = AtomicI->getOrdering();
331 AtomicI->setOrdering(NewOrdering);
332 return bracketInstWithFences(I: AtomicI, Order: FenceOrdering);
333 }
334 if (!ShouldInsertFences)
335 return tryInsertTrailingSeqCstFence(AtomicI);
336 return false;
337}
338
339/// In a single-threaded environment, atomic operations can be lowered to their
340/// non-atomic equivalents: fences are removed, and atomic loads, stores, RMW,
341/// and cmpxchg become plain memory operations.
342bool AtomicExpandImpl::lowerToNonAtomic(Instruction *I) {
343 if (auto *FI = dyn_cast<FenceInst>(Val: I)) {
344 FI->eraseFromParent();
345 return true;
346 }
347
348 if (auto *CXI = dyn_cast<AtomicCmpXchgInst>(Val: I))
349 return lowerAtomicCmpXchgInst(CXI);
350
351 if (auto *RMWI = dyn_cast<AtomicRMWInst>(Val: I))
352 return lowerAtomicRMWInst(RMWI);
353
354 if (auto *LI = dyn_cast<LoadInst>(Val: I)) {
355 if (LI->isAtomic()) {
356 LI->setAtomic(Ordering: AtomicOrdering::NotAtomic);
357 LI->setElementwise(false);
358 return true;
359 }
360
361 return false;
362 }
363
364 if (auto *SI = dyn_cast<StoreInst>(Val: I)) {
365 if (SI->isAtomic()) {
366 SI->setAtomic(Ordering: AtomicOrdering::NotAtomic);
367 SI->setElementwise(false);
368 return true;
369 }
370
371 return false;
372 }
373
374 return false;
375}
376
377bool AtomicExpandImpl::processAtomicInstr(Instruction *I) {
378 if (SingleThreaded)
379 return lowerToNonAtomic(I);
380
381 if (auto *LI = dyn_cast<LoadInst>(Val: I)) {
382 if (!LI->isAtomic())
383 return false;
384
385 if (!atomicSizeSupported(TLI, I: LI)) {
386 expandAtomicLoadToLibcall(LI);
387 return true;
388 }
389
390 bool MadeChange = false;
391 if (TLI->shouldCastAtomicLoadInIR(LI) ==
392 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
393 LI = convertAtomicLoadToIntegerType(LI);
394 MadeChange = true;
395 }
396
397 MadeChange |= tryInsertFencesForAtomic(
398 AtomicI: LI, OrderingRequiresFence: isAcquireOrStronger(AO: LI->getOrdering()), NewOrdering: AtomicOrdering::Monotonic);
399
400 MadeChange |= tryExpandAtomicLoad(LI);
401 return MadeChange;
402 }
403
404 if (auto *SI = dyn_cast<StoreInst>(Val: I)) {
405 if (!SI->isAtomic())
406 return false;
407
408 if (!atomicSizeSupported(TLI, I: SI)) {
409 expandAtomicStoreToLibcall(LI: SI);
410 return true;
411 }
412
413 bool MadeChange = false;
414 if (TLI->shouldCastAtomicStoreInIR(SI) ==
415 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
416 SI = convertAtomicStoreToIntegerType(SI);
417 MadeChange = true;
418 }
419
420 MadeChange |= tryInsertFencesForAtomic(
421 AtomicI: SI, OrderingRequiresFence: isReleaseOrStronger(AO: SI->getOrdering()), NewOrdering: AtomicOrdering::Monotonic);
422
423 MadeChange |= tryExpandAtomicStore(SI);
424 return MadeChange;
425 }
426
427 if (auto *RMWI = dyn_cast<AtomicRMWInst>(Val: I)) {
428 if (!atomicSizeSupported(TLI, I: RMWI)) {
429 expandAtomicRMWToLibcall(I: RMWI);
430 return true;
431 }
432
433 bool MadeChange = false;
434 if (TLI->shouldCastAtomicRMWIInIR(RMWI) ==
435 TargetLoweringBase::AtomicExpansionKind::CastToInteger) {
436 RMWI = convertAtomicXchgToIntegerType(RMWI);
437 MadeChange = true;
438 }
439
440 MadeChange |= tryInsertFencesForAtomic(
441 AtomicI: RMWI,
442 OrderingRequiresFence: isReleaseOrStronger(AO: RMWI->getOrdering()) ||
443 isAcquireOrStronger(AO: RMWI->getOrdering()),
444 NewOrdering: TLI->atomicOperationOrderAfterFenceSplit(I: RMWI));
445
446 // There are two different ways of expanding RMW instructions:
447 // - into a load if it is idempotent
448 // - into a Cmpxchg/LL-SC loop otherwise
449 // we try them in that order.
450 MadeChange |= (isIdempotentRMW(RMWI) && simplifyIdempotentRMW(RMWI)) ||
451 tryExpandAtomicRMW(AI: RMWI);
452 return MadeChange;
453 }
454
455 if (auto *CASI = dyn_cast<AtomicCmpXchgInst>(Val: I)) {
456 if (!atomicSizeSupported(TLI, I: CASI)) {
457 expandAtomicCASToLibcall(I: CASI);
458 return true;
459 }
460
461 // TODO: when we're ready to make the change at the IR level, we can
462 // extend convertCmpXchgToInteger for floating point too.
463 bool MadeChange = false;
464 if (CASI->getCompareOperand()->getType()->isPointerTy()) {
465 // TODO: add a TLI hook to control this so that each target can
466 // convert to lowering the original type one at a time.
467 CASI = convertCmpXchgToIntegerType(CI: CASI);
468 MadeChange = true;
469 }
470
471 auto CmpXchgExpansion = TLI->shouldExpandAtomicCmpXchgInIR(AI: CASI);
472 if (TLI->shouldInsertFencesForAtomic(I: CASI)) {
473 if (CmpXchgExpansion == TargetLoweringBase::AtomicExpansionKind::None &&
474 (isReleaseOrStronger(AO: CASI->getSuccessOrdering()) ||
475 isAcquireOrStronger(AO: CASI->getSuccessOrdering()) ||
476 isAcquireOrStronger(AO: CASI->getFailureOrdering()))) {
477 // If a compare and swap is lowered to LL/SC, we can do smarter fence
478 // insertion, with a stronger one on the success path than on the
479 // failure path. As a result, fence insertion is directly done by
480 // expandAtomicCmpXchg in that case.
481 AtomicOrdering FenceOrdering = CASI->getMergedOrdering();
482 AtomicOrdering CASOrdering =
483 TLI->atomicOperationOrderAfterFenceSplit(I: CASI);
484 CASI->setSuccessOrdering(CASOrdering);
485 CASI->setFailureOrdering(CASOrdering);
486 MadeChange |= bracketInstWithFences(I: CASI, Order: FenceOrdering);
487 }
488 } else if (CmpXchgExpansion !=
489 TargetLoweringBase::AtomicExpansionKind::LLSC) {
490 // CmpXchg LLSC is handled in expandAtomicCmpXchg().
491 MadeChange |= tryInsertTrailingSeqCstFence(AtomicI: CASI);
492 }
493
494 MadeChange |= tryExpandAtomicCmpXchg(CI: CASI);
495 return MadeChange;
496 }
497
498 return false;
499}
500
501bool AtomicExpandImpl::run(Function &F,
502 const ModuleLibcallLoweringInfo &LibcallResult,
503 const TargetMachine *TM) {
504 SingleThreaded = F.getParent()->getThreadModel() == ThreadModel::Single;
505
506 const auto *Subtarget = TM->getSubtargetImpl(F);
507 // In a single-threaded environment atomics are lowered to non-atomic form
508 if (!SingleThreaded && !Subtarget->enableAtomicExpand())
509 return false;
510 TLI = Subtarget->getTargetLowering();
511 LibcallLowering = &getLibcallLowering(ModuleInfo: LibcallResult, Subtarget: *Subtarget);
512 DL = &F.getDataLayout();
513
514 bool MadeChange = false;
515
516 for (Function::iterator BBI = F.begin(), BBE = F.end(); BBI != BBE; ++BBI) {
517 BasicBlock *BB = &*BBI;
518
519 BasicBlock::reverse_iterator Next;
520
521 for (BasicBlock::reverse_iterator I = BB->rbegin(), E = BB->rend(); I != E;
522 I = Next) {
523 Instruction &Inst = *I;
524 Next = std::next(x: I);
525
526 if (processAtomicInstr(I: &Inst)) {
527 MadeChange = true;
528
529 // New blocks may have been inserted.
530 BBE = F.end();
531 }
532 }
533 }
534
535 return MadeChange;
536}
537
538bool AtomicExpandLegacy::runOnFunction(Function &F) {
539
540 auto *TPC = getAnalysisIfAvailable<TargetPassConfig>();
541 if (!TPC)
542 return false;
543 auto *TM = &TPC->getTM<TargetMachine>();
544
545 const ModuleLibcallLoweringInfo &LibcallResult =
546 getAnalysis<LibcallLoweringInfoWrapper>().getResult(M: *F.getParent());
547 AtomicExpandImpl AE;
548 return AE.run(F, LibcallResult, TM);
549}
550
551FunctionPass *llvm::createAtomicExpandLegacyPass() {
552 return new AtomicExpandLegacy();
553}
554
555PreservedAnalyses AtomicExpandPass::run(Function &F,
556 FunctionAnalysisManager &FAM) {
557 auto &MAMProxy = FAM.getResult<ModuleAnalysisManagerFunctionProxy>(IR&: F);
558
559 const ModuleLibcallLoweringInfo *LibcallResult =
560 MAMProxy.getCachedResult<LibcallLoweringModuleAnalysis>(IR&: *F.getParent());
561
562 if (!LibcallResult) {
563 F.getContext().emitError(ErrorStr: "'" + LibcallLoweringModuleAnalysis::name() +
564 "' analysis required");
565 return PreservedAnalyses::all();
566 }
567
568 AtomicExpandImpl AE;
569
570 bool Changed = AE.run(F, LibcallResult: *LibcallResult, TM);
571 if (!Changed)
572 return PreservedAnalyses::all();
573
574 return PreservedAnalyses::none();
575}
576
577bool AtomicExpandImpl::bracketInstWithFences(Instruction *I,
578 AtomicOrdering Order) {
579 ReplacementIRBuilder Builder(I, *DL);
580
581 auto LeadingFence = TLI->emitLeadingFence(Builder, Inst: I, Ord: Order);
582
583 auto TrailingFence = TLI->emitTrailingFence(Builder, Inst: I, Ord: Order);
584 // We have a guard here because not every atomic operation generates a
585 // trailing fence.
586 if (TrailingFence)
587 TrailingFence->moveAfter(MovePos: I);
588
589 return (LeadingFence || TrailingFence);
590}
591
592/// Get the iX type with the same bitwidth as T.
593IntegerType *
594AtomicExpandImpl::getCorrespondingIntegerType(Type *T, const DataLayout &DL) {
595 EVT VT = TLI->getMemValueType(DL, Ty: T);
596 unsigned BitWidth = VT.getStoreSizeInBits();
597 assert(BitWidth == VT.getSizeInBits() && "must be a power of two");
598 return IntegerType::get(C&: T->getContext(), NumBits: BitWidth);
599}
600
601/// Convert an atomic load of a non-integral type to an integer load of the
602/// equivalent bitwidth. See the function comment on
603/// convertAtomicStoreToIntegerType for background.
604LoadInst *AtomicExpandImpl::convertAtomicLoadToIntegerType(LoadInst *LI) {
605 auto *M = LI->getModule();
606 Type *NewTy = getCorrespondingIntegerType(T: LI->getType(), DL: M->getDataLayout());
607
608 ReplacementIRBuilder Builder(LI, *DL);
609
610 Value *Addr = LI->getPointerOperand();
611
612 auto *NewLI = Builder.CreateLoad(Ty: NewTy, Ptr: Addr, Props: LI->getProperties());
613 LLVM_DEBUG(dbgs() << "Replaced " << *LI << " with " << *NewLI << "\n");
614
615 Value *NewVal = LI->getType()->isPtrOrPtrVectorTy()
616 ? Builder.CreateIntToPtr(V: NewLI, DestTy: LI->getType())
617 : Builder.CreateBitCast(V: NewLI, DestTy: LI->getType());
618 LI->replaceAllUsesWith(V: NewVal);
619 LI->eraseFromParent();
620 return NewLI;
621}
622
623AtomicRMWInst *
624AtomicExpandImpl::convertAtomicXchgToIntegerType(AtomicRMWInst *RMWI) {
625 assert(RMWI->getOperation() == AtomicRMWInst::Xchg);
626
627 auto *M = RMWI->getModule();
628 Type *NewTy =
629 getCorrespondingIntegerType(T: RMWI->getType(), DL: M->getDataLayout());
630
631 ReplacementIRBuilder Builder(RMWI, *DL);
632
633 Value *Addr = RMWI->getPointerOperand();
634 Value *Val = RMWI->getValOperand();
635 Value *NewVal = Builder.CreateBitPreservingCastChain(DL: *DL, V: Val, NewTy);
636
637 auto *NewRMWI = Builder.CreateAtomicRMW(Op: AtomicRMWInst::Xchg, Ptr: Addr, Val: NewVal,
638 Align: RMWI->getAlign(), Ordering: RMWI->getOrdering(),
639 SSID: RMWI->getSyncScopeID());
640 NewRMWI->setVolatile(RMWI->isVolatile());
641 copyMetadataForAtomic(Dest&: *NewRMWI, Source: *RMWI);
642 LLVM_DEBUG(dbgs() << "Replaced " << *RMWI << " with " << *NewRMWI << "\n");
643
644 Value *NewRVal =
645 Builder.CreateBitPreservingCastChain(DL: *DL, V: NewRMWI, NewTy: RMWI->getType());
646 RMWI->replaceAllUsesWith(V: NewRVal);
647 RMWI->eraseFromParent();
648 return NewRMWI;
649}
650
651bool AtomicExpandImpl::tryExpandAtomicLoad(LoadInst *LI) {
652 switch (TLI->shouldExpandAtomicLoadInIR(LI)) {
653 case TargetLoweringBase::AtomicExpansionKind::None:
654 return false;
655 case TargetLoweringBase::AtomicExpansionKind::LLSC:
656 expandAtomicOpToLLSC(
657 I: LI, ResultTy: LI->getType(), Addr: LI->getPointerOperand(), AddrAlign: LI->getAlign(),
658 MemOpOrder: LI->getOrdering(),
659 PerformOp: [](IRBuilderBase &Builder, Value *Loaded) { return Loaded; });
660 return true;
661 case TargetLoweringBase::AtomicExpansionKind::LLOnly:
662 return expandAtomicLoadToLL(LI);
663 case TargetLoweringBase::AtomicExpansionKind::CmpXChg:
664 return expandAtomicLoadToCmpXchg(LI);
665 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
666 LI->setAtomic(Ordering: AtomicOrdering::NotAtomic);
667 return true;
668 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
669 TLI->emitExpandAtomicLoad(LI);
670 return true;
671 default:
672 llvm_unreachable("Unhandled case in tryExpandAtomicLoad");
673 }
674}
675
676bool AtomicExpandImpl::tryExpandAtomicStore(StoreInst *SI) {
677 switch (TLI->shouldExpandAtomicStoreInIR(SI)) {
678 case TargetLoweringBase::AtomicExpansionKind::None:
679 return false;
680 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
681 TLI->emitExpandAtomicStore(SI);
682 return true;
683 case TargetLoweringBase::AtomicExpansionKind::Expand:
684 expandAtomicStoreToXChg(SI);
685 return true;
686 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
687 SI->setAtomic(Ordering: AtomicOrdering::NotAtomic);
688 return true;
689 default:
690 llvm_unreachable("Unhandled case in tryExpandAtomicStore");
691 }
692}
693
694bool AtomicExpandImpl::expandAtomicLoadToLL(LoadInst *LI) {
695 ReplacementIRBuilder Builder(LI, *DL);
696
697 // On some architectures, load-linked instructions are atomic for larger
698 // sizes than normal loads. For example, the only 64-bit load guaranteed
699 // to be single-copy atomic by ARM is an ldrexd (A3.5.3).
700 Value *Val = TLI->emitLoadLinked(Builder, ValueTy: LI->getType(),
701 Addr: LI->getPointerOperand(), Ord: LI->getOrdering());
702 TLI->emitAtomicCmpXchgNoStoreLLBalance(Builder);
703
704 LI->replaceAllUsesWith(V: Val);
705 LI->eraseFromParent();
706
707 return true;
708}
709
710bool AtomicExpandImpl::expandAtomicLoadToCmpXchg(LoadInst *LI) {
711 ReplacementIRBuilder Builder(LI, *DL);
712 AtomicOrdering Order = LI->getOrdering();
713 if (Order == AtomicOrdering::Unordered)
714 Order = AtomicOrdering::Monotonic;
715
716 Value *Addr = LI->getPointerOperand();
717 Type *Ty = LI->getType();
718
719 // cmpxchg supports only integer and pointer operands. If the load type is
720 // FP or vector, run the cmpxchg on the same-sized integer and bitcast the
721 // result back; mirrors createCmpXchgInstFun.
722 bool NeedBitcast = Ty->isFloatingPointTy() || Ty->isVectorTy();
723 Type *CmpXchgTy = Ty;
724 if (NeedBitcast)
725 CmpXchgTy = Builder.getIntNTy(N: Ty->getPrimitiveSizeInBits());
726 Constant *DummyVal = Constant::getNullValue(Ty: CmpXchgTy);
727
728 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
729 Ptr: Addr, Cmp: DummyVal, New: DummyVal, Align: LI->getAlign(), SuccessOrdering: Order,
730 FailureOrdering: AtomicCmpXchgInst::getStrongestFailureOrdering(SuccessOrdering: Order),
731 SSID: LI->getSyncScopeID());
732 Pair->setVolatile(LI->isVolatile());
733 Value *Loaded = Builder.CreateExtractValue(Agg: Pair, Idxs: 0, Name: "loaded");
734 if (NeedBitcast)
735 Loaded = Builder.CreateBitCast(V: Loaded, DestTy: Ty);
736
737 LI->replaceAllUsesWith(V: Loaded);
738 LI->eraseFromParent();
739
740 return true;
741}
742
743/// Convert an atomic store of a non-integral type to an integer store of the
744/// equivalent bitwidth. We used to not support floating point or vector
745/// atomics in the IR at all. The backends learned to deal with the bitcast
746/// idiom because that was the only way of expressing the notion of a atomic
747/// float or vector store. The long term plan is to teach each backend to
748/// instruction select from the original atomic store, but as a migration
749/// mechanism, we convert back to the old format which the backends understand.
750/// Each backend will need individual work to recognize the new format.
751StoreInst *AtomicExpandImpl::convertAtomicStoreToIntegerType(StoreInst *SI) {
752 ReplacementIRBuilder Builder(SI, *DL);
753 auto *M = SI->getModule();
754 Type *NewTy = getCorrespondingIntegerType(T: SI->getValueOperand()->getType(),
755 DL: M->getDataLayout());
756 Value *NewVal = SI->getValueOperand()->getType()->isPtrOrPtrVectorTy()
757 ? Builder.CreatePtrToInt(V: SI->getValueOperand(), DestTy: NewTy)
758 : Builder.CreateBitCast(V: SI->getValueOperand(), DestTy: NewTy);
759
760 Value *Addr = SI->getPointerOperand();
761
762 StoreInst *NewSI = Builder.CreateStore(Val: NewVal, Ptr: Addr, Props: SI->getProperties());
763 copyMetadataForAtomic(Dest&: *NewSI, Source: *SI);
764 LLVM_DEBUG(dbgs() << "Replaced " << *SI << " with " << *NewSI << "\n");
765 SI->eraseFromParent();
766 return NewSI;
767}
768
769void AtomicExpandImpl::expandAtomicStoreToXChg(StoreInst *SI) {
770 // This function is only called on atomic stores that are too large to be
771 // atomic if implemented as a native store. So we replace them by an
772 // atomic swap, that can be implemented for example as a ldrex/strex on ARM
773 // or lock cmpxchg8/16b on X86, as these are atomic for larger sizes.
774 // It is the responsibility of the target to only signal expansion via
775 // shouldExpandAtomicRMW in cases where this is required and possible.
776 ReplacementIRBuilder Builder(SI, *DL);
777 AtomicOrdering Ordering = SI->getOrdering();
778 assert(Ordering != AtomicOrdering::NotAtomic);
779 AtomicOrdering RMWOrdering = Ordering == AtomicOrdering::Unordered
780 ? AtomicOrdering::Monotonic
781 : Ordering;
782 AtomicRMWInst *AI = Builder.CreateAtomicRMW(
783 Op: AtomicRMWInst::Xchg, Ptr: SI->getPointerOperand(), Val: SI->getValueOperand(),
784 Align: SI->getAlign(), Ordering: RMWOrdering, SSID: SI->getSyncScopeID());
785 AI->setVolatile(SI->isVolatile());
786 SI->eraseFromParent();
787
788 // Now we have an appropriate swap instruction, lower it as usual.
789 tryExpandAtomicRMW(AI);
790}
791
792static void createCmpXchgInstFun(IRBuilderBase &Builder, Value *Addr,
793 Value *Loaded, Value *NewVal, Align AddrAlign,
794 AtomicOrdering MemOpOrder, SyncScope::ID SSID,
795 bool IsVolatile, Value *&Success,
796 Value *&NewLoaded, Instruction *MetadataSrc) {
797 Type *OrigTy = NewVal->getType();
798
799 // This code can go away when cmpxchg supports FP and vector types.
800 assert(!OrigTy->isPointerTy());
801 bool NeedBitcast = OrigTy->isFloatingPointTy() || OrigTy->isVectorTy();
802 if (NeedBitcast) {
803 IntegerType *IntTy = Builder.getIntNTy(N: OrigTy->getPrimitiveSizeInBits());
804 NewVal = Builder.CreateBitCast(V: NewVal, DestTy: IntTy);
805 Loaded = Builder.CreateBitCast(V: Loaded, DestTy: IntTy);
806 }
807
808 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
809 Ptr: Addr, Cmp: Loaded, New: NewVal, Align: AddrAlign, SuccessOrdering: MemOpOrder,
810 FailureOrdering: AtomicCmpXchgInst::getStrongestFailureOrdering(SuccessOrdering: MemOpOrder), SSID);
811 Pair->setVolatile(IsVolatile);
812 if (MetadataSrc)
813 copyMetadataForAtomic(Dest&: *Pair, Source: *MetadataSrc);
814
815 Success = Builder.CreateExtractValue(Agg: Pair, Idxs: 1, Name: "success");
816 NewLoaded = Builder.CreateExtractValue(Agg: Pair, Idxs: 0, Name: "newloaded");
817
818 if (NeedBitcast)
819 NewLoaded = Builder.CreateBitCast(V: NewLoaded, DestTy: OrigTy);
820}
821
822void AtomicExpandImpl::expandAtomicSubToAdd(AtomicRMWInst *AI) {
823 ReplacementIRBuilder Builder(AI, *DL);
824 AtomicRMWInst::BinOp NewOp;
825 Value *NewVal;
826
827 switch (AI->getOperation()) {
828 case AtomicRMWInst::Sub:
829 NewOp = AtomicRMWInst::Add;
830 NewVal = Builder.CreateNeg(V: AI->getValOperand(), Name: "neg");
831 break;
832 case AtomicRMWInst::FSub:
833 NewOp = AtomicRMWInst::FAdd;
834 NewVal = Builder.CreateFNeg(V: AI->getValOperand(), Name: "fneg");
835 break;
836 default:
837 llvm_unreachable("unsupported atomicrmw expansion");
838 }
839
840 AI->setOperation(NewOp);
841 AI->setOperand(i_nocapture: 1, Val_nocapture: NewVal);
842}
843
844bool AtomicExpandImpl::tryExpandAtomicRMW(AtomicRMWInst *AI) {
845 LLVMContext &Ctx = AI->getModule()->getContext();
846 TargetLowering::AtomicExpansionKind Kind = TLI->shouldExpandAtomicRMWInIR(RMW: AI);
847 switch (Kind) {
848 case TargetLoweringBase::AtomicExpansionKind::None:
849 return false;
850 case TargetLoweringBase::AtomicExpansionKind::LLSC: {
851 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
852 unsigned ValueSize = getAtomicOpSize(RMWI: AI);
853 if (ValueSize < MinCASSize) {
854 expandPartwordAtomicRMW(I: AI,
855 ExpansionKind: TargetLoweringBase::AtomicExpansionKind::LLSC);
856 } else {
857 auto PerformOp = [&](IRBuilderBase &Builder, Value *Loaded) {
858 return buildAtomicRMWValue(Op: AI->getOperation(), Builder, Loaded,
859 Val: AI->getValOperand());
860 };
861 expandAtomicOpToLLSC(I: AI, ResultTy: AI->getType(), Addr: AI->getPointerOperand(),
862 AddrAlign: AI->getAlign(), MemOpOrder: AI->getOrdering(), PerformOp);
863 }
864 return true;
865 }
866 case TargetLoweringBase::AtomicExpansionKind::CmpXChg: {
867 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
868 unsigned ValueSize = getAtomicOpSize(RMWI: AI);
869 if (ValueSize < MinCASSize) {
870 expandPartwordAtomicRMW(I: AI,
871 ExpansionKind: TargetLoweringBase::AtomicExpansionKind::CmpXChg);
872 } else {
873 SmallVector<StringRef> SSNs;
874 Ctx.getSyncScopeNames(SSNs);
875 auto MemScope = SSNs[AI->getSyncScopeID()].empty()
876 ? "system"
877 : SSNs[AI->getSyncScopeID()];
878 OptimizationRemarkEmitter ORE(AI->getFunction());
879 ORE.emit(RemarkBuilder: [&]() {
880 return OptimizationRemark(DEBUG_TYPE, "Passed", AI)
881 << "A compare and swap loop was generated for an atomic "
882 << AI->getOperationName(Op: AI->getOperation()) << " operation at "
883 << MemScope << " memory scope";
884 });
885 expandAtomicRMWToCmpXchg(AI, CreateCmpXchg: createCmpXchgInstFun);
886 }
887 return true;
888 }
889 case TargetLoweringBase::AtomicExpansionKind::MaskedIntrinsic: {
890 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
891 unsigned ValueSize = getAtomicOpSize(RMWI: AI);
892 if (ValueSize < MinCASSize) {
893 AtomicRMWInst::BinOp Op = AI->getOperation();
894 // Widen And/Or/Xor and give the target another chance at expanding it.
895 if (Op == AtomicRMWInst::Or || Op == AtomicRMWInst::Xor ||
896 Op == AtomicRMWInst::And) {
897 tryExpandAtomicRMW(AI: widenPartwordAtomicRMW(AI));
898 return true;
899 }
900 }
901 expandAtomicRMWToMaskedIntrinsic(AI);
902 return true;
903 }
904 case TargetLoweringBase::AtomicExpansionKind::BitTestIntrinsic: {
905 TLI->emitBitTestAtomicRMWIntrinsic(AI);
906 return true;
907 }
908 case TargetLoweringBase::AtomicExpansionKind::CmpArithIntrinsic: {
909 TLI->emitCmpArithAtomicRMWIntrinsic(AI);
910 return true;
911 }
912 case TargetLoweringBase::AtomicExpansionKind::Expand:
913 expandAtomicSubToAdd(AI);
914 return true;
915 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
916 return lowerAtomicRMWInst(RMWI: AI);
917 case TargetLoweringBase::AtomicExpansionKind::CustomExpand:
918 TLI->emitExpandAtomicRMW(AI);
919 return true;
920 default:
921 llvm_unreachable("Unhandled case in tryExpandAtomicRMW");
922 }
923}
924
925namespace {
926
927struct PartwordMaskValues {
928 // These three fields are guaranteed to be set by createMaskInstrs.
929 Type *WordType = nullptr;
930 Type *ValueType = nullptr;
931 Type *IntValueType = nullptr;
932 Value *AlignedAddr = nullptr;
933 Align AlignedAddrAlignment;
934 // The remaining fields can be null.
935 Value *ShiftAmt = nullptr;
936 Value *Mask = nullptr;
937 Value *Inv_Mask = nullptr;
938};
939
940[[maybe_unused]]
941raw_ostream &operator<<(raw_ostream &O, const PartwordMaskValues &PMV) {
942 auto PrintObj = [&O](auto *V) {
943 if (V)
944 O << *V;
945 else
946 O << "nullptr";
947 O << '\n';
948 };
949 O << "PartwordMaskValues {\n";
950 O << " WordType: ";
951 PrintObj(PMV.WordType);
952 O << " ValueType: ";
953 PrintObj(PMV.ValueType);
954 O << " AlignedAddr: ";
955 PrintObj(PMV.AlignedAddr);
956 O << " AlignedAddrAlignment: " << PMV.AlignedAddrAlignment.value() << '\n';
957 O << " ShiftAmt: ";
958 PrintObj(PMV.ShiftAmt);
959 O << " Mask: ";
960 PrintObj(PMV.Mask);
961 O << " Inv_Mask: ";
962 PrintObj(PMV.Inv_Mask);
963 O << "}\n";
964 return O;
965}
966
967} // end anonymous namespace
968
969/// This is a helper function which builds instructions to provide
970/// values necessary for partword atomic operations. It takes an
971/// incoming address, Addr, and ValueType, and constructs the address,
972/// shift-amounts and masks needed to work with a larger value of size
973/// WordSize.
974///
975/// AlignedAddr: Addr rounded down to a multiple of WordSize
976///
977/// ShiftAmt: Number of bits to right-shift a WordSize value loaded
978/// from AlignAddr for it to have the same value as if
979/// ValueType was loaded from Addr.
980///
981/// Mask: Value to mask with the value loaded from AlignAddr to
982/// include only the part that would've been loaded from Addr.
983///
984/// Inv_Mask: The inverse of Mask.
985static PartwordMaskValues createMaskInstrs(IRBuilderBase &Builder,
986 Instruction *I, Type *ValueType,
987 Value *Addr, Align AddrAlign,
988 unsigned MinWordSize) {
989 PartwordMaskValues PMV;
990
991 Module *M = I->getModule();
992 LLVMContext &Ctx = M->getContext();
993 const DataLayout &DL = M->getDataLayout();
994 unsigned ValueSize = DL.getTypeStoreSize(Ty: ValueType);
995
996 PMV.ValueType = PMV.IntValueType = ValueType;
997 if (PMV.ValueType->isFloatingPointTy() || PMV.ValueType->isVectorTy())
998 PMV.IntValueType =
999 Type::getIntNTy(C&: Ctx, N: ValueType->getPrimitiveSizeInBits());
1000
1001 PMV.WordType = MinWordSize > ValueSize ? Type::getIntNTy(C&: Ctx, N: MinWordSize * 8)
1002 : ValueType;
1003 if (PMV.ValueType == PMV.WordType) {
1004 PMV.AlignedAddr = Addr;
1005 PMV.AlignedAddrAlignment = AddrAlign;
1006 PMV.ShiftAmt = ConstantInt::get(Ty: PMV.ValueType, V: 0);
1007 PMV.Mask = ConstantInt::get(Ty: PMV.ValueType, V: ~0, /*isSigned*/ IsSigned: true);
1008 return PMV;
1009 }
1010
1011 PMV.AlignedAddrAlignment = Align(MinWordSize);
1012
1013 assert(ValueSize < MinWordSize);
1014
1015 PointerType *PtrTy = cast<PointerType>(Val: Addr->getType());
1016 IntegerType *IntTy = DL.getIndexType(C&: Ctx, AddressSpace: PtrTy->getAddressSpace());
1017 Value *PtrLSB;
1018
1019 if (AddrAlign < MinWordSize) {
1020 PMV.AlignedAddr = Builder.CreateIntrinsic(
1021 ID: Intrinsic::ptrmask, OverloadTypes: {PtrTy, IntTy},
1022 Args: {Addr, ConstantInt::getSigned(Ty: IntTy, V: ~(uint64_t)(MinWordSize - 1))},
1023 FMFSource: nullptr, Name: "AlignedAddr");
1024
1025 Value *AddrInt = Builder.CreatePtrToInt(V: Addr, DestTy: IntTy);
1026 PtrLSB = Builder.CreateAnd(LHS: AddrInt, RHS: MinWordSize - 1, Name: "PtrLSB");
1027 } else {
1028 // If the alignment is high enough, the LSB are known 0.
1029 PMV.AlignedAddr = Addr;
1030 PtrLSB = ConstantInt::getNullValue(Ty: IntTy);
1031 }
1032
1033 if (DL.isLittleEndian()) {
1034 // turn bytes into bits
1035 PMV.ShiftAmt = Builder.CreateShl(LHS: PtrLSB, RHS: 3);
1036 } else {
1037 // turn bytes into bits, and count from the other side.
1038 PMV.ShiftAmt = Builder.CreateShl(
1039 LHS: Builder.CreateXor(LHS: PtrLSB, RHS: MinWordSize - ValueSize), RHS: 3);
1040 }
1041
1042 PMV.ShiftAmt = Builder.CreateTrunc(V: PMV.ShiftAmt, DestTy: PMV.WordType, Name: "ShiftAmt");
1043 PMV.Mask = Builder.CreateShl(
1044 LHS: ConstantInt::get(Ty: PMV.WordType, V: (1 << (ValueSize * 8)) - 1), RHS: PMV.ShiftAmt,
1045 Name: "Mask");
1046
1047 PMV.Inv_Mask = Builder.CreateNot(V: PMV.Mask, Name: "Inv_Mask");
1048
1049 return PMV;
1050}
1051
1052static Value *extractMaskedValue(IRBuilderBase &Builder, Value *WideWord,
1053 const PartwordMaskValues &PMV) {
1054 assert(WideWord->getType() == PMV.WordType && "Widened type mismatch");
1055 if (PMV.WordType == PMV.ValueType)
1056 return WideWord;
1057
1058 Value *Shift = Builder.CreateLShr(LHS: WideWord, RHS: PMV.ShiftAmt, Name: "shifted");
1059 Value *Trunc = Builder.CreateTrunc(V: Shift, DestTy: PMV.IntValueType, Name: "extracted");
1060 return Builder.CreateBitCast(V: Trunc, DestTy: PMV.ValueType);
1061}
1062
1063static Value *insertMaskedValue(IRBuilderBase &Builder, Value *WideWord,
1064 Value *Updated, const PartwordMaskValues &PMV) {
1065 assert(WideWord->getType() == PMV.WordType && "Widened type mismatch");
1066 assert(Updated->getType() == PMV.ValueType && "Value type mismatch");
1067 if (PMV.WordType == PMV.ValueType)
1068 return Updated;
1069
1070 Updated = Builder.CreateBitCast(V: Updated, DestTy: PMV.IntValueType);
1071
1072 Value *ZExt = Builder.CreateZExt(V: Updated, DestTy: PMV.WordType, Name: "extended");
1073 Value *Shift =
1074 Builder.CreateShl(LHS: ZExt, RHS: PMV.ShiftAmt, Name: "shifted", /*HasNUW*/ true);
1075 Value *And = Builder.CreateAnd(LHS: WideWord, RHS: PMV.Inv_Mask, Name: "unmasked");
1076 Value *Or = Builder.CreateOr(LHS: And, RHS: Shift, Name: "inserted");
1077 return Or;
1078}
1079
1080/// Emit IR to implement a masked version of a given atomicrmw
1081/// operation. (That is, only the bits under the Mask should be
1082/// affected by the operation)
1083static Value *performMaskedAtomicOp(AtomicRMWInst::BinOp Op,
1084 IRBuilderBase &Builder, Value *Loaded,
1085 Value *ValOperand_Shifted, Value *Inc,
1086 const PartwordMaskValues &PMV) {
1087 // TODO: update to use
1088 // https://graphics.stanford.edu/~seander/bithacks.html#MaskedMerge in order
1089 // to merge bits from two values without requiring PMV.Inv_Mask.
1090
1091 assert(Op != AtomicRMWInst::Or && Op != AtomicRMWInst::Xor &&
1092 Op != AtomicRMWInst::And &&
1093 "Or/Xor/And handled by widenPartwordAtomicRMW");
1094
1095 if (Op == AtomicRMWInst::Xchg) {
1096 // Clear all the bits we are exchanging out. These are the bits under the
1097 // mask. We can clear them with an `and` of the inverse mask.
1098 Value *Loaded_MaskOut = Builder.CreateAnd(LHS: Loaded, RHS: PMV.Inv_Mask);
1099 // Now that the prevous bits are cleared, we can swap in the new value with
1100 // an `or`.
1101 Value *FinalVal = Builder.CreateOr(LHS: Loaded_MaskOut, RHS: ValOperand_Shifted);
1102 return FinalVal;
1103 }
1104
1105 if (Op == AtomicRMWInst::Nand ||
1106 (!PMV.ValueType->isVectorTy() &&
1107 (Op == AtomicRMWInst::Add || Op == AtomicRMWInst::Sub))) {
1108 // For `Nand` and non-vector `Add` and `Sub`, we can perform the operation
1109 // on the entire word because the extra bits in the unmasked region don't
1110 // affect the computation in the masked region. The operation might still
1111 // overwrite the unmasked region (e.g. from integer overflow or underflow),
1112 // so we have to reapply the unmasked region afterwards.
1113 //
1114 // This trick doesn't work for vector `Add` and `Sub` because we use a
1115 // scalar operation on the entire word. Scalarizing vector `Add` and `Sub`
1116 // isn't legal because the vector versions may have element-wise overflows.
1117 // TODO: For these, can we use a wider vector op with additional lanes?
1118
1119 // Atomic operation across the entire word.
1120 Value *NewVal =
1121 buildAtomicRMWValue(Op, Builder, Loaded, Val: ValOperand_Shifted);
1122 // Reapply the bits in the unmasked region.
1123 Value *NewVal_Masked = Builder.CreateAnd(LHS: NewVal, RHS: PMV.Mask);
1124 Value *Loaded_MaskOut = Builder.CreateAnd(LHS: Loaded, RHS: PMV.Inv_Mask);
1125 Value *FinalVal = Builder.CreateOr(LHS: Loaded_MaskOut, RHS: NewVal_Masked);
1126 return FinalVal;
1127 }
1128
1129 // All other ops operate on the sub-word size. Truncate down to the
1130 // original size, and expand out again after doing the operation. Bitcasts
1131 // will be inserted for FP values.
1132 assert(!ValOperand_Shifted);
1133 Value *Loaded_Extract = extractMaskedValue(Builder, WideWord: Loaded, PMV);
1134 Value *NewVal = buildAtomicRMWValue(Op, Builder, Loaded: Loaded_Extract, Val: Inc);
1135 Value *FinalVal = insertMaskedValue(Builder, WideWord: Loaded, Updated: NewVal, PMV);
1136 return FinalVal;
1137}
1138
1139/// Expand a sub-word atomicrmw operation into an appropriate
1140/// word-sized operation.
1141///
1142/// It will create an LL/SC or cmpxchg loop, as appropriate, the same
1143/// way as a typical atomicrmw expansion. The only difference here is
1144/// that the operation inside of the loop may operate upon only a
1145/// part of the value.
1146void AtomicExpandImpl::expandPartwordAtomicRMW(
1147 AtomicRMWInst *AI, TargetLoweringBase::AtomicExpansionKind ExpansionKind) {
1148 // Widen And/Or/Xor and give the target another chance at expanding it.
1149 AtomicRMWInst::BinOp Op = AI->getOperation();
1150 if (Op == AtomicRMWInst::Or || Op == AtomicRMWInst::Xor ||
1151 Op == AtomicRMWInst::And) {
1152 tryExpandAtomicRMW(AI: widenPartwordAtomicRMW(AI));
1153 return;
1154 }
1155 AtomicOrdering MemOpOrder = AI->getOrdering();
1156 SyncScope::ID SSID = AI->getSyncScopeID();
1157
1158 ReplacementIRBuilder Builder(AI, *DL);
1159
1160 PartwordMaskValues PMV =
1161 createMaskInstrs(Builder, I: AI, ValueType: AI->getType(), Addr: AI->getPointerOperand(),
1162 AddrAlign: AI->getAlign(), MinWordSize: TLI->getMinCmpXchgSizeInBits() / 8);
1163
1164 Value *ValOperand_Shifted = nullptr;
1165 bool NeedsShiftedOperand =
1166 Op == AtomicRMWInst::Xchg || Op == AtomicRMWInst::Nand ||
1167 (!PMV.ValueType->isVectorTy() &&
1168 (Op == AtomicRMWInst::Add || Op == AtomicRMWInst::Sub));
1169
1170 if (NeedsShiftedOperand) {
1171 Value *ValOp = Builder.CreateBitCast(V: AI->getValOperand(), DestTy: PMV.IntValueType);
1172 ValOperand_Shifted =
1173 Builder.CreateShl(LHS: Builder.CreateZExt(V: ValOp, DestTy: PMV.WordType), RHS: PMV.ShiftAmt,
1174 Name: "ValOperand_Shifted");
1175 }
1176
1177 auto PerformPartwordOp = [&](IRBuilderBase &Builder, Value *Loaded) {
1178 return performMaskedAtomicOp(Op, Builder, Loaded, ValOperand_Shifted,
1179 Inc: AI->getValOperand(), PMV);
1180 };
1181
1182 Value *OldResult;
1183 if (ExpansionKind == TargetLoweringBase::AtomicExpansionKind::CmpXChg) {
1184 OldResult = insertRMWCmpXchgLoop(Builder, ResultType: PMV.WordType, Addr: PMV.AlignedAddr,
1185 AddrAlign: PMV.AlignedAddrAlignment, MemOpOrder, SSID,
1186 IsVolatile: AI->isVolatile(), PerformOp: PerformPartwordOp,
1187 CreateCmpXchg: createCmpXchgInstFun, MetadataSrc: AI);
1188 } else {
1189 assert(ExpansionKind == TargetLoweringBase::AtomicExpansionKind::LLSC);
1190 OldResult = insertRMWLLSCLoop(Builder, ResultTy: PMV.WordType, Addr: PMV.AlignedAddr,
1191 AddrAlign: PMV.AlignedAddrAlignment, MemOpOrder,
1192 PerformOp: PerformPartwordOp);
1193 }
1194
1195 Value *FinalOldResult = extractMaskedValue(Builder, WideWord: OldResult, PMV);
1196 AI->replaceAllUsesWith(V: FinalOldResult);
1197 AI->eraseFromParent();
1198}
1199
1200// Widen the bitwise atomicrmw (or/xor/and) to the minimum supported width.
1201AtomicRMWInst *AtomicExpandImpl::widenPartwordAtomicRMW(AtomicRMWInst *AI) {
1202 ReplacementIRBuilder Builder(AI, *DL);
1203 AtomicRMWInst::BinOp Op = AI->getOperation();
1204
1205 assert((Op == AtomicRMWInst::Or || Op == AtomicRMWInst::Xor ||
1206 Op == AtomicRMWInst::And) &&
1207 "Unable to widen operation");
1208
1209 PartwordMaskValues PMV =
1210 createMaskInstrs(Builder, I: AI, ValueType: AI->getType(), Addr: AI->getPointerOperand(),
1211 AddrAlign: AI->getAlign(), MinWordSize: TLI->getMinCmpXchgSizeInBits() / 8);
1212
1213 Value *ValOp = AI->getValOperand();
1214 if (ValOp->getType()->isVectorTy())
1215 // For vectors, bitcast to the integer type before extending. Note that
1216 // or/xor/and on vectors are equivalent to the same operation on an integer
1217 // that spans the vector, so we can use the integer type for the operation.
1218 ValOp = Builder.CreateBitCast(V: ValOp, DestTy: PMV.IntValueType);
1219 Value *ValOperand_Shifted =
1220 Builder.CreateShl(LHS: Builder.CreateZExt(V: ValOp, DestTy: PMV.WordType), RHS: PMV.ShiftAmt,
1221 Name: "ValOperand_Shifted");
1222
1223 Value *NewOperand;
1224
1225 if (Op == AtomicRMWInst::And)
1226 NewOperand =
1227 Builder.CreateOr(LHS: ValOperand_Shifted, RHS: PMV.Inv_Mask, Name: "AndOperand");
1228 else
1229 NewOperand = ValOperand_Shifted;
1230
1231 AtomicRMWInst *NewAI = Builder.CreateAtomicRMW(
1232 Op, Ptr: PMV.AlignedAddr, Val: NewOperand, Align: PMV.AlignedAddrAlignment,
1233 Ordering: AI->getOrdering(), SSID: AI->getSyncScopeID());
1234
1235 NewAI->setVolatile(AI->isVolatile());
1236 copyMetadataForAtomic(Dest&: *NewAI, Source: *AI);
1237
1238 Value *FinalOldResult = extractMaskedValue(Builder, WideWord: NewAI, PMV);
1239 AI->replaceAllUsesWith(V: FinalOldResult);
1240 AI->eraseFromParent();
1241 return NewAI;
1242}
1243
1244bool AtomicExpandImpl::expandPartwordCmpXchg(AtomicCmpXchgInst *CI) {
1245 // The basic idea here is that we're expanding a cmpxchg of a
1246 // smaller memory size up to a word-sized cmpxchg. To do this, we
1247 // need to add a retry-loop for strong cmpxchg, so that
1248 // modifications to other parts of the word don't cause a spurious
1249 // failure.
1250
1251 // This generates code like the following:
1252 // [[Setup mask values PMV.*]]
1253 // %NewVal_Shifted = shl i32 %NewVal, %PMV.ShiftAmt
1254 // %Cmp_Shifted = shl i32 %Cmp, %PMV.ShiftAmt
1255 // %InitLoaded = load i32* %addr
1256 // %InitLoaded_MaskOut = and i32 %InitLoaded, %PMV.Inv_Mask
1257 // br partword.cmpxchg.loop
1258 // partword.cmpxchg.loop:
1259 // %Loaded_MaskOut = phi i32 [ %InitLoaded_MaskOut, %entry ],
1260 // [ %OldVal_MaskOut, %partword.cmpxchg.failure ]
1261 // %FullWord_NewVal = or i32 %Loaded_MaskOut, %NewVal_Shifted
1262 // %FullWord_Cmp = or i32 %Loaded_MaskOut, %Cmp_Shifted
1263 // %NewCI = cmpxchg i32* %PMV.AlignedAddr, i32 %FullWord_Cmp,
1264 // i32 %FullWord_NewVal success_ordering failure_ordering
1265 // %OldVal = extractvalue { i32, i1 } %NewCI, 0
1266 // %Success = extractvalue { i32, i1 } %NewCI, 1
1267 // br i1 %Success, label %partword.cmpxchg.end,
1268 // label %partword.cmpxchg.failure
1269 // partword.cmpxchg.failure:
1270 // %OldVal_MaskOut = and i32 %OldVal, %PMV.Inv_Mask
1271 // %ShouldContinue = icmp ne i32 %Loaded_MaskOut, %OldVal_MaskOut
1272 // br i1 %ShouldContinue, label %partword.cmpxchg.loop,
1273 // label %partword.cmpxchg.end
1274 // partword.cmpxchg.end:
1275 // %tmp1 = lshr i32 %OldVal, %PMV.ShiftAmt
1276 // %FinalOldVal = trunc i32 %tmp1 to i8
1277 // %tmp2 = insertvalue { i8, i1 } undef, i8 %FinalOldVal, 0
1278 // %Res = insertvalue { i8, i1 } %25, i1 %Success, 1
1279
1280 Value *Addr = CI->getPointerOperand();
1281 Value *Cmp = CI->getCompareOperand();
1282 Value *NewVal = CI->getNewValOperand();
1283
1284 BasicBlock *BB = CI->getParent();
1285 Function *F = BB->getParent();
1286 ReplacementIRBuilder Builder(CI, *DL);
1287 LLVMContext &Ctx = Builder.getContext();
1288
1289 BasicBlock *EndBB =
1290 BB->splitBasicBlock(I: CI->getIterator(), BBName: "partword.cmpxchg.end");
1291 auto FailureBB =
1292 BasicBlock::Create(Context&: Ctx, Name: "partword.cmpxchg.failure", Parent: F, InsertBefore: EndBB);
1293 auto LoopBB = BasicBlock::Create(Context&: Ctx, Name: "partword.cmpxchg.loop", Parent: F, InsertBefore: FailureBB);
1294
1295 // The split call above "helpfully" added a branch at the end of BB
1296 // (to the wrong place).
1297 std::prev(x: BB->end())->eraseFromParent();
1298 Builder.SetInsertPoint(BB);
1299
1300 PartwordMaskValues PMV =
1301 createMaskInstrs(Builder, I: CI, ValueType: CI->getCompareOperand()->getType(), Addr,
1302 AddrAlign: CI->getAlign(), MinWordSize: TLI->getMinCmpXchgSizeInBits() / 8);
1303
1304 // Shift the incoming values over, into the right location in the word.
1305 Value *NewVal_Shifted =
1306 Builder.CreateShl(LHS: Builder.CreateZExt(V: NewVal, DestTy: PMV.WordType), RHS: PMV.ShiftAmt);
1307 Value *Cmp_Shifted =
1308 Builder.CreateShl(LHS: Builder.CreateZExt(V: Cmp, DestTy: PMV.WordType), RHS: PMV.ShiftAmt);
1309
1310 // Load the entire current word, and mask into place the expected and new
1311 // values
1312 LoadInst *InitLoaded = Builder.CreateLoad(Ty: PMV.WordType, Ptr: PMV.AlignedAddr);
1313 Value *InitLoaded_MaskOut = Builder.CreateAnd(LHS: InitLoaded, RHS: PMV.Inv_Mask);
1314 Builder.CreateBr(Dest: LoopBB);
1315
1316 // partword.cmpxchg.loop:
1317 Builder.SetInsertPoint(LoopBB);
1318 PHINode *Loaded_MaskOut = Builder.CreatePHI(Ty: PMV.WordType, NumReservedValues: 2);
1319 Loaded_MaskOut->addIncoming(V: InitLoaded_MaskOut, BB);
1320
1321 // The initial load must be atomic with the same synchronization scope
1322 // to avoid a data race with concurrent stores. If the instruction being
1323 // emulated is volatile, issue a volatile load.
1324 // addIncoming is done first so that any replaceAllUsesWith calls during
1325 // normalization correctly update the PHI incoming value.
1326 InitLoaded->setVolatile(CI->isVolatile());
1327 if (TLI->shouldIssueAtomicLoadForAtomicEmulationLoop()) {
1328 InitLoaded->setAtomic(Ordering: AtomicOrdering::Monotonic, SSID: CI->getSyncScopeID());
1329 // The newly created load might need to be lowered further. Because it is
1330 // created in the same block as the atomicrmw, the AtomicExpand loop will
1331 // not process it again.
1332 processAtomicInstr(I: InitLoaded);
1333 }
1334
1335 // Mask/Or the expected and new values into place in the loaded word.
1336 Value *FullWord_NewVal = Builder.CreateOr(LHS: Loaded_MaskOut, RHS: NewVal_Shifted);
1337 Value *FullWord_Cmp = Builder.CreateOr(LHS: Loaded_MaskOut, RHS: Cmp_Shifted);
1338 AtomicCmpXchgInst *NewCI = Builder.CreateAtomicCmpXchg(
1339 Ptr: PMV.AlignedAddr, Cmp: FullWord_Cmp, New: FullWord_NewVal, Align: PMV.AlignedAddrAlignment,
1340 SuccessOrdering: CI->getSuccessOrdering(), FailureOrdering: CI->getFailureOrdering(), SSID: CI->getSyncScopeID());
1341 NewCI->setVolatile(CI->isVolatile());
1342 // When we're building a strong cmpxchg, we need a loop, so you
1343 // might think we could use a weak cmpxchg inside. But, using strong
1344 // allows the below comparison for ShouldContinue, and we're
1345 // expecting the underlying cmpxchg to be a machine instruction,
1346 // which is strong anyways.
1347 NewCI->setWeak(CI->isWeak());
1348
1349 Value *OldVal = Builder.CreateExtractValue(Agg: NewCI, Idxs: 0);
1350 Value *Success = Builder.CreateExtractValue(Agg: NewCI, Idxs: 1);
1351
1352 if (CI->isWeak())
1353 Builder.CreateBr(Dest: EndBB);
1354 else
1355 Builder.CreateCondBr(Cond: Success, True: EndBB, False: FailureBB);
1356
1357 // partword.cmpxchg.failure:
1358 Builder.SetInsertPoint(FailureBB);
1359 // Upon failure, verify that the masked-out part of the loaded value
1360 // has been modified. If it didn't, abort the cmpxchg, since the
1361 // masked-in part must've.
1362 Value *OldVal_MaskOut = Builder.CreateAnd(LHS: OldVal, RHS: PMV.Inv_Mask);
1363 Value *ShouldContinue = Builder.CreateICmpNE(LHS: Loaded_MaskOut, RHS: OldVal_MaskOut);
1364 Builder.CreateCondBr(Cond: ShouldContinue, True: LoopBB, False: EndBB);
1365
1366 // Add the second value to the phi from above
1367 Loaded_MaskOut->addIncoming(V: OldVal_MaskOut, BB: FailureBB);
1368
1369 // partword.cmpxchg.end:
1370 Builder.SetInsertPoint(CI);
1371
1372 Value *FinalOldVal = extractMaskedValue(Builder, WideWord: OldVal, PMV);
1373 Value *Res = PoisonValue::get(T: CI->getType());
1374 Res = Builder.CreateInsertValue(Agg: Res, Val: FinalOldVal, Idxs: 0);
1375 Res = Builder.CreateInsertValue(Agg: Res, Val: Success, Idxs: 1);
1376
1377 CI->replaceAllUsesWith(V: Res);
1378 CI->eraseFromParent();
1379 return true;
1380}
1381
1382void AtomicExpandImpl::expandAtomicOpToLLSC(
1383 Instruction *I, Type *ResultType, Value *Addr, Align AddrAlign,
1384 AtomicOrdering MemOpOrder,
1385 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp) {
1386 ReplacementIRBuilder Builder(I, *DL);
1387 Value *Loaded = insertRMWLLSCLoop(Builder, ResultTy: ResultType, Addr, AddrAlign,
1388 MemOpOrder, PerformOp);
1389
1390 I->replaceAllUsesWith(V: Loaded);
1391 I->eraseFromParent();
1392}
1393
1394void AtomicExpandImpl::expandAtomicRMWToMaskedIntrinsic(AtomicRMWInst *AI) {
1395 ReplacementIRBuilder Builder(AI, *DL);
1396
1397 PartwordMaskValues PMV =
1398 createMaskInstrs(Builder, I: AI, ValueType: AI->getType(), Addr: AI->getPointerOperand(),
1399 AddrAlign: AI->getAlign(), MinWordSize: TLI->getMinCmpXchgSizeInBits() / 8);
1400
1401 // The value operand must be sign-extended for signed min/max so that the
1402 // target's signed comparison instructions can be used. Otherwise, just
1403 // zero-ext.
1404 Instruction::CastOps CastOp = Instruction::ZExt;
1405 AtomicRMWInst::BinOp RMWOp = AI->getOperation();
1406 if (RMWOp == AtomicRMWInst::Max || RMWOp == AtomicRMWInst::Min)
1407 CastOp = Instruction::SExt;
1408
1409 Value *ValOperand_Shifted = Builder.CreateShl(
1410 LHS: Builder.CreateCast(Op: CastOp, V: AI->getValOperand(), DestTy: PMV.WordType),
1411 RHS: PMV.ShiftAmt, Name: "ValOperand_Shifted");
1412 Value *OldResult = TLI->emitMaskedAtomicRMWIntrinsic(
1413 Builder, AI, AlignedAddr: PMV.AlignedAddr, Incr: ValOperand_Shifted, Mask: PMV.Mask, ShiftAmt: PMV.ShiftAmt,
1414 Ord: AI->getOrdering());
1415 Value *FinalOldResult = extractMaskedValue(Builder, WideWord: OldResult, PMV);
1416 AI->replaceAllUsesWith(V: FinalOldResult);
1417 AI->eraseFromParent();
1418}
1419
1420void AtomicExpandImpl::expandAtomicCmpXchgToMaskedIntrinsic(
1421 AtomicCmpXchgInst *CI) {
1422 ReplacementIRBuilder Builder(CI, *DL);
1423
1424 PartwordMaskValues PMV = createMaskInstrs(
1425 Builder, I: CI, ValueType: CI->getCompareOperand()->getType(), Addr: CI->getPointerOperand(),
1426 AddrAlign: CI->getAlign(), MinWordSize: TLI->getMinCmpXchgSizeInBits() / 8);
1427
1428 Value *CmpVal_Shifted = Builder.CreateShl(
1429 LHS: Builder.CreateZExt(V: CI->getCompareOperand(), DestTy: PMV.WordType), RHS: PMV.ShiftAmt,
1430 Name: "CmpVal_Shifted");
1431 Value *NewVal_Shifted = Builder.CreateShl(
1432 LHS: Builder.CreateZExt(V: CI->getNewValOperand(), DestTy: PMV.WordType), RHS: PMV.ShiftAmt,
1433 Name: "NewVal_Shifted");
1434 Value *OldVal = TLI->emitMaskedAtomicCmpXchgIntrinsic(
1435 Builder, CI, AlignedAddr: PMV.AlignedAddr, CmpVal: CmpVal_Shifted, NewVal: NewVal_Shifted, Mask: PMV.Mask,
1436 Ord: CI->getMergedOrdering());
1437 Value *FinalOldVal = extractMaskedValue(Builder, WideWord: OldVal, PMV);
1438 Value *Res = PoisonValue::get(T: CI->getType());
1439 Res = Builder.CreateInsertValue(Agg: Res, Val: FinalOldVal, Idxs: 0);
1440 Value *Success = Builder.CreateICmpEQ(
1441 LHS: CmpVal_Shifted, RHS: Builder.CreateAnd(LHS: OldVal, RHS: PMV.Mask), Name: "Success");
1442 Res = Builder.CreateInsertValue(Agg: Res, Val: Success, Idxs: 1);
1443
1444 CI->replaceAllUsesWith(V: Res);
1445 CI->eraseFromParent();
1446}
1447
1448Value *AtomicExpandImpl::insertRMWLLSCLoop(
1449 IRBuilderBase &Builder, Type *ResultTy, Value *Addr, Align AddrAlign,
1450 AtomicOrdering MemOpOrder,
1451 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp) {
1452 LLVMContext &Ctx = Builder.getContext();
1453 BasicBlock *BB = Builder.GetInsertBlock();
1454 Function *F = BB->getParent();
1455
1456 assert(AddrAlign >= F->getDataLayout().getTypeStoreSize(ResultTy) &&
1457 "Expected at least natural alignment at this point.");
1458
1459 // Given: atomicrmw some_op iN* %addr, iN %incr ordering
1460 //
1461 // The standard expansion we produce is:
1462 // [...]
1463 // atomicrmw.start:
1464 // %loaded = @load.linked(%addr)
1465 // %new = some_op iN %loaded, %incr
1466 // %stored = @store_conditional(%new, %addr)
1467 // %try_again = icmp i32 ne %stored, 0
1468 // br i1 %try_again, label %loop, label %atomicrmw.end
1469 // atomicrmw.end:
1470 // [...]
1471 BasicBlock *ExitBB =
1472 BB->splitBasicBlock(I: Builder.GetInsertPoint(), BBName: "atomicrmw.end");
1473 BasicBlock *LoopBB = BasicBlock::Create(Context&: Ctx, Name: "atomicrmw.start", Parent: F, InsertBefore: ExitBB);
1474
1475 // The split call above "helpfully" added a branch at the end of BB (to the
1476 // wrong place).
1477 std::prev(x: BB->end())->eraseFromParent();
1478 Builder.SetInsertPoint(BB);
1479 Builder.CreateBr(Dest: LoopBB);
1480
1481 // Start the main loop block now that we've taken care of the preliminaries.
1482 Builder.SetInsertPoint(LoopBB);
1483 Value *Loaded = TLI->emitLoadLinked(Builder, ValueTy: ResultTy, Addr, Ord: MemOpOrder);
1484
1485 Value *NewVal = PerformOp(Builder, Loaded);
1486
1487 Value *StoreSuccess =
1488 TLI->emitStoreConditional(Builder, Val: NewVal, Addr, Ord: MemOpOrder);
1489 Value *TryAgain = Builder.CreateICmpNE(
1490 LHS: StoreSuccess, RHS: ConstantInt::get(Ty: IntegerType::get(C&: Ctx, NumBits: 32), V: 0), Name: "tryagain");
1491
1492 Instruction *CondBr = Builder.CreateCondBr(Cond: TryAgain, True: LoopBB, False: ExitBB);
1493
1494 // Atomic RMW expands to a Load-linked / Store-Conditional loop, because it is
1495 // hard to predict precise branch weigths we mark the branch as "unknown"
1496 // (50/50) to prevent misleading optimizations.
1497 setExplicitlyUnknownBranchWeightsIfProfiled(I&: *CondBr, DEBUG_TYPE);
1498
1499 Builder.SetInsertPoint(ExitBB->begin());
1500 return Loaded;
1501}
1502
1503/// Convert an atomic cmpxchg of a non-integral type to an integer cmpxchg of
1504/// the equivalent bitwidth. We used to not support pointer cmpxchg in the
1505/// IR. As a migration step, we convert back to what use to be the standard
1506/// way to represent a pointer cmpxchg so that we can update backends one by
1507/// one.
1508AtomicCmpXchgInst *
1509AtomicExpandImpl::convertCmpXchgToIntegerType(AtomicCmpXchgInst *CI) {
1510 auto *M = CI->getModule();
1511 Type *NewTy = getCorrespondingIntegerType(T: CI->getCompareOperand()->getType(),
1512 DL: M->getDataLayout());
1513
1514 ReplacementIRBuilder Builder(CI, *DL);
1515
1516 Value *Addr = CI->getPointerOperand();
1517
1518 Value *NewCmp = Builder.CreatePtrToInt(V: CI->getCompareOperand(), DestTy: NewTy);
1519 Value *NewNewVal = Builder.CreatePtrToInt(V: CI->getNewValOperand(), DestTy: NewTy);
1520
1521 auto *NewCI = Builder.CreateAtomicCmpXchg(
1522 Ptr: Addr, Cmp: NewCmp, New: NewNewVal, Align: CI->getAlign(), SuccessOrdering: CI->getSuccessOrdering(),
1523 FailureOrdering: CI->getFailureOrdering(), SSID: CI->getSyncScopeID());
1524 NewCI->setVolatile(CI->isVolatile());
1525 NewCI->setWeak(CI->isWeak());
1526 LLVM_DEBUG(dbgs() << "Replaced " << *CI << " with " << *NewCI << "\n");
1527
1528 Value *OldVal = Builder.CreateExtractValue(Agg: NewCI, Idxs: 0);
1529 Value *Succ = Builder.CreateExtractValue(Agg: NewCI, Idxs: 1);
1530
1531 OldVal = Builder.CreateIntToPtr(V: OldVal, DestTy: CI->getCompareOperand()->getType());
1532
1533 Value *Res = PoisonValue::get(T: CI->getType());
1534 Res = Builder.CreateInsertValue(Agg: Res, Val: OldVal, Idxs: 0);
1535 Res = Builder.CreateInsertValue(Agg: Res, Val: Succ, Idxs: 1);
1536
1537 CI->replaceAllUsesWith(V: Res);
1538 CI->eraseFromParent();
1539 return NewCI;
1540}
1541
1542bool AtomicExpandImpl::expandAtomicCmpXchg(AtomicCmpXchgInst *CI) {
1543 AtomicOrdering SuccessOrder = CI->getSuccessOrdering();
1544 AtomicOrdering FailureOrder = CI->getFailureOrdering();
1545 Value *Addr = CI->getPointerOperand();
1546 BasicBlock *BB = CI->getParent();
1547 Function *F = BB->getParent();
1548 LLVMContext &Ctx = F->getContext();
1549 // If shouldInsertFencesForAtomic() returns true, then the target does not
1550 // want to deal with memory orders, and emitLeading/TrailingFence should take
1551 // care of everything. Otherwise, emitLeading/TrailingFence are no-op and we
1552 // should preserve the ordering.
1553 bool ShouldInsertFencesForAtomic = TLI->shouldInsertFencesForAtomic(I: CI);
1554 AtomicOrdering MemOpOrder = ShouldInsertFencesForAtomic
1555 ? AtomicOrdering::Monotonic
1556 : CI->getMergedOrdering();
1557
1558 // In implementations which use a barrier to achieve release semantics, we can
1559 // delay emitting this barrier until we know a store is actually going to be
1560 // attempted. The cost of this delay is that we need 2 copies of the block
1561 // emitting the load-linked, affecting code size.
1562 //
1563 // Ideally, this logic would be unconditional except for the minsize check
1564 // since in other cases the extra blocks naturally collapse down to the
1565 // minimal loop. Unfortunately, this puts too much stress on later
1566 // optimisations so we avoid emitting the extra logic in those cases too.
1567 bool HasReleasedLoadBB = !CI->isWeak() && ShouldInsertFencesForAtomic &&
1568 SuccessOrder != AtomicOrdering::Monotonic &&
1569 SuccessOrder != AtomicOrdering::Acquire &&
1570 !F->hasMinSize();
1571
1572 // There's no overhead for sinking the release barrier in a weak cmpxchg, so
1573 // do it even on minsize.
1574 bool UseUnconditionalReleaseBarrier = F->hasMinSize() && !CI->isWeak();
1575
1576 // Given: cmpxchg some_op iN* %addr, iN %desired, iN %new success_ord fail_ord
1577 //
1578 // The full expansion we produce is:
1579 // [...]
1580 // %aligned.addr = ...
1581 // cmpxchg.start:
1582 // %unreleasedload = @load.linked(%aligned.addr)
1583 // %unreleasedload.extract = extract value from %unreleasedload
1584 // %should_store = icmp eq %unreleasedload.extract, %desired
1585 // br i1 %should_store, label %cmpxchg.releasingstore,
1586 // label %cmpxchg.nostore
1587 // cmpxchg.releasingstore:
1588 // fence?
1589 // br label cmpxchg.trystore
1590 // cmpxchg.trystore:
1591 // %loaded.trystore = phi [%unreleasedload, %cmpxchg.releasingstore],
1592 // [%releasedload, %cmpxchg.releasedload]
1593 // %updated.new = insert %new into %loaded.trystore
1594 // %stored = @store_conditional(%updated.new, %aligned.addr)
1595 // %success = icmp eq i32 %stored, 0
1596 // br i1 %success, label %cmpxchg.success,
1597 // label %cmpxchg.releasedload/%cmpxchg.failure
1598 // cmpxchg.releasedload:
1599 // %releasedload = @load.linked(%aligned.addr)
1600 // %releasedload.extract = extract value from %releasedload
1601 // %should_store = icmp eq %releasedload.extract, %desired
1602 // br i1 %should_store, label %cmpxchg.trystore,
1603 // label %cmpxchg.failure
1604 // cmpxchg.success:
1605 // fence?
1606 // br label %cmpxchg.end
1607 // cmpxchg.nostore:
1608 // %loaded.nostore = phi [%unreleasedload, %cmpxchg.start],
1609 // [%releasedload,
1610 // %cmpxchg.releasedload/%cmpxchg.trystore]
1611 // @load_linked_fail_balance()?
1612 // br label %cmpxchg.failure
1613 // cmpxchg.failure:
1614 // fence?
1615 // br label %cmpxchg.end
1616 // cmpxchg.end:
1617 // %loaded.exit = phi [%loaded.nostore, %cmpxchg.failure],
1618 // [%loaded.trystore, %cmpxchg.trystore]
1619 // %success = phi i1 [true, %cmpxchg.success], [false, %cmpxchg.failure]
1620 // %loaded = extract value from %loaded.exit
1621 // %restmp = insertvalue { iN, i1 } undef, iN %loaded, 0
1622 // %res = insertvalue { iN, i1 } %restmp, i1 %success, 1
1623 // [...]
1624 BasicBlock *ExitBB = BB->splitBasicBlock(I: CI->getIterator(), BBName: "cmpxchg.end");
1625 auto FailureBB = BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.failure", Parent: F, InsertBefore: ExitBB);
1626 auto NoStoreBB = BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.nostore", Parent: F, InsertBefore: FailureBB);
1627 auto SuccessBB = BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.success", Parent: F, InsertBefore: NoStoreBB);
1628 auto ReleasedLoadBB =
1629 BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.releasedload", Parent: F, InsertBefore: SuccessBB);
1630 auto TryStoreBB =
1631 BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.trystore", Parent: F, InsertBefore: ReleasedLoadBB);
1632 auto ReleasingStoreBB =
1633 BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.fencedstore", Parent: F, InsertBefore: TryStoreBB);
1634 auto StartBB = BasicBlock::Create(Context&: Ctx, Name: "cmpxchg.start", Parent: F, InsertBefore: ReleasingStoreBB);
1635
1636 ReplacementIRBuilder Builder(CI, *DL);
1637
1638 // The split call above "helpfully" added a branch at the end of BB (to the
1639 // wrong place), but we might want a fence too. It's easiest to just remove
1640 // the branch entirely.
1641 std::prev(x: BB->end())->eraseFromParent();
1642 Builder.SetInsertPoint(BB);
1643 if (ShouldInsertFencesForAtomic && UseUnconditionalReleaseBarrier)
1644 TLI->emitLeadingFence(Builder, Inst: CI, Ord: SuccessOrder);
1645
1646 PartwordMaskValues PMV =
1647 createMaskInstrs(Builder, I: CI, ValueType: CI->getCompareOperand()->getType(), Addr,
1648 AddrAlign: CI->getAlign(), MinWordSize: TLI->getMinCmpXchgSizeInBits() / 8);
1649 Builder.CreateBr(Dest: StartBB);
1650
1651 // Start the main loop block now that we've taken care of the preliminaries.
1652 Builder.SetInsertPoint(StartBB);
1653 Value *UnreleasedLoad =
1654 TLI->emitLoadLinked(Builder, ValueTy: PMV.WordType, Addr: PMV.AlignedAddr, Ord: MemOpOrder);
1655 Value *UnreleasedLoadExtract =
1656 extractMaskedValue(Builder, WideWord: UnreleasedLoad, PMV);
1657 Value *ShouldStore = Builder.CreateICmpEQ(
1658 LHS: UnreleasedLoadExtract, RHS: CI->getCompareOperand(), Name: "should_store");
1659
1660 // If the cmpxchg doesn't actually need any ordering when it fails, we can
1661 // jump straight past that fence instruction (if it exists).
1662 Builder.CreateCondBr(Cond: ShouldStore, True: ReleasingStoreBB, False: NoStoreBB,
1663 BranchWeights: MDBuilder(F->getContext()).createLikelyBranchWeights());
1664
1665 Builder.SetInsertPoint(ReleasingStoreBB);
1666 if (ShouldInsertFencesForAtomic && !UseUnconditionalReleaseBarrier)
1667 TLI->emitLeadingFence(Builder, Inst: CI, Ord: SuccessOrder);
1668 Builder.CreateBr(Dest: TryStoreBB);
1669
1670 Builder.SetInsertPoint(TryStoreBB);
1671 PHINode *LoadedTryStore =
1672 Builder.CreatePHI(Ty: PMV.WordType, NumReservedValues: 2, Name: "loaded.trystore");
1673 LoadedTryStore->addIncoming(V: UnreleasedLoad, BB: ReleasingStoreBB);
1674 Value *NewValueInsert =
1675 insertMaskedValue(Builder, WideWord: LoadedTryStore, Updated: CI->getNewValOperand(), PMV);
1676 Value *StoreSuccess = TLI->emitStoreConditional(Builder, Val: NewValueInsert,
1677 Addr: PMV.AlignedAddr, Ord: MemOpOrder);
1678 StoreSuccess = Builder.CreateICmpEQ(
1679 LHS: StoreSuccess, RHS: ConstantInt::get(Ty: Type::getInt32Ty(C&: Ctx), V: 0), Name: "success");
1680 BasicBlock *RetryBB = HasReleasedLoadBB ? ReleasedLoadBB : StartBB;
1681 Builder.CreateCondBr(Cond: StoreSuccess, True: SuccessBB,
1682 False: CI->isWeak() ? FailureBB : RetryBB,
1683 BranchWeights: MDBuilder(F->getContext()).createLikelyBranchWeights());
1684
1685 Builder.SetInsertPoint(ReleasedLoadBB);
1686 Value *SecondLoad;
1687 if (HasReleasedLoadBB) {
1688 SecondLoad =
1689 TLI->emitLoadLinked(Builder, ValueTy: PMV.WordType, Addr: PMV.AlignedAddr, Ord: MemOpOrder);
1690 Value *SecondLoadExtract = extractMaskedValue(Builder, WideWord: SecondLoad, PMV);
1691 ShouldStore = Builder.CreateICmpEQ(LHS: SecondLoadExtract,
1692 RHS: CI->getCompareOperand(), Name: "should_store");
1693
1694 // If the cmpxchg doesn't actually need any ordering when it fails, we can
1695 // jump straight past that fence instruction (if it exists).
1696 Builder.CreateCondBr(
1697 Cond: ShouldStore, True: TryStoreBB, False: NoStoreBB,
1698 BranchWeights: MDBuilder(F->getContext()).createLikelyBranchWeights());
1699 // Update PHI node in TryStoreBB.
1700 LoadedTryStore->addIncoming(V: SecondLoad, BB: ReleasedLoadBB);
1701 } else
1702 Builder.CreateUnreachable();
1703
1704 // Make sure later instructions don't get reordered with a fence if
1705 // necessary.
1706 Builder.SetInsertPoint(SuccessBB);
1707 if (ShouldInsertFencesForAtomic ||
1708 TLI->shouldInsertTrailingSeqCstFenceForAtomicStore(I: CI))
1709 TLI->emitTrailingFence(Builder, Inst: CI, Ord: SuccessOrder);
1710 Builder.CreateBr(Dest: ExitBB);
1711
1712 Builder.SetInsertPoint(NoStoreBB);
1713 PHINode *LoadedNoStore =
1714 Builder.CreatePHI(Ty: UnreleasedLoad->getType(), NumReservedValues: 2, Name: "loaded.nostore");
1715 LoadedNoStore->addIncoming(V: UnreleasedLoad, BB: StartBB);
1716 if (HasReleasedLoadBB)
1717 LoadedNoStore->addIncoming(V: SecondLoad, BB: ReleasedLoadBB);
1718
1719 // In the failing case, where we don't execute the store-conditional, the
1720 // target might want to balance out the load-linked with a dedicated
1721 // instruction (e.g., on ARM, clearing the exclusive monitor).
1722 TLI->emitAtomicCmpXchgNoStoreLLBalance(Builder);
1723 Builder.CreateBr(Dest: FailureBB);
1724
1725 Builder.SetInsertPoint(FailureBB);
1726 PHINode *LoadedFailure =
1727 Builder.CreatePHI(Ty: UnreleasedLoad->getType(), NumReservedValues: 2, Name: "loaded.failure");
1728 LoadedFailure->addIncoming(V: LoadedNoStore, BB: NoStoreBB);
1729 if (CI->isWeak())
1730 LoadedFailure->addIncoming(V: LoadedTryStore, BB: TryStoreBB);
1731 if (ShouldInsertFencesForAtomic)
1732 TLI->emitTrailingFence(Builder, Inst: CI, Ord: FailureOrder);
1733 Builder.CreateBr(Dest: ExitBB);
1734
1735 // Finally, we have control-flow based knowledge of whether the cmpxchg
1736 // succeeded or not. We expose this to later passes by converting any
1737 // subsequent "icmp eq/ne %loaded, %oldval" into a use of an appropriate
1738 // PHI.
1739 Builder.SetInsertPoint(ExitBB->begin());
1740 PHINode *LoadedExit =
1741 Builder.CreatePHI(Ty: UnreleasedLoad->getType(), NumReservedValues: 2, Name: "loaded.exit");
1742 LoadedExit->addIncoming(V: LoadedTryStore, BB: SuccessBB);
1743 LoadedExit->addIncoming(V: LoadedFailure, BB: FailureBB);
1744 PHINode *Success = Builder.CreatePHI(Ty: Type::getInt1Ty(C&: Ctx), NumReservedValues: 2, Name: "success");
1745 Success->addIncoming(V: ConstantInt::getTrue(Context&: Ctx), BB: SuccessBB);
1746 Success->addIncoming(V: ConstantInt::getFalse(Context&: Ctx), BB: FailureBB);
1747
1748 // This is the "exit value" from the cmpxchg expansion. It may be of
1749 // a type wider than the one in the cmpxchg instruction.
1750 Value *LoadedFull = LoadedExit;
1751
1752 Builder.SetInsertPoint(std::next(x: Success->getIterator()));
1753 Value *Loaded = extractMaskedValue(Builder, WideWord: LoadedFull, PMV);
1754
1755 // Look for any users of the cmpxchg that are just comparing the loaded value
1756 // against the desired one, and replace them with the CFG-derived version.
1757 SmallVector<ExtractValueInst *, 2> PrunedInsts;
1758 for (auto *User : CI->users()) {
1759 ExtractValueInst *EV = dyn_cast<ExtractValueInst>(Val: User);
1760 if (!EV)
1761 continue;
1762
1763 assert(EV->getNumIndices() == 1 && EV->getIndices()[0] <= 1 &&
1764 "weird extraction from { iN, i1 }");
1765
1766 if (EV->getIndices()[0] == 0)
1767 EV->replaceAllUsesWith(V: Loaded);
1768 else
1769 EV->replaceAllUsesWith(V: Success);
1770
1771 PrunedInsts.push_back(Elt: EV);
1772 }
1773
1774 // We can remove the instructions now we're no longer iterating through them.
1775 for (auto *EV : PrunedInsts)
1776 EV->eraseFromParent();
1777
1778 if (!CI->use_empty()) {
1779 // Some use of the full struct return that we don't understand has happened,
1780 // so we've got to reconstruct it properly.
1781 Value *Res;
1782 Res = Builder.CreateInsertValue(Agg: PoisonValue::get(T: CI->getType()), Val: Loaded, Idxs: 0);
1783 Res = Builder.CreateInsertValue(Agg: Res, Val: Success, Idxs: 1);
1784
1785 CI->replaceAllUsesWith(V: Res);
1786 }
1787
1788 CI->eraseFromParent();
1789 return true;
1790}
1791
1792bool AtomicExpandImpl::isIdempotentRMW(AtomicRMWInst *RMWI) {
1793 if (RMWI->isVolatile())
1794 return false;
1795 // TODO: Add floating point support.
1796 auto C = dyn_cast<ConstantInt>(Val: RMWI->getValOperand());
1797 if (!C)
1798 return false;
1799
1800 switch (RMWI->getOperation()) {
1801 case AtomicRMWInst::Add:
1802 case AtomicRMWInst::Sub:
1803 case AtomicRMWInst::Or:
1804 case AtomicRMWInst::Xor:
1805 return C->isZero();
1806 case AtomicRMWInst::And:
1807 return C->isMinusOne();
1808 case AtomicRMWInst::Min:
1809 return C->isMaxValue(IsSigned: true);
1810 case AtomicRMWInst::Max:
1811 return C->isMinValue(IsSigned: true);
1812 case AtomicRMWInst::UMin:
1813 return C->isMaxValue(IsSigned: false);
1814 case AtomicRMWInst::UMax:
1815 return C->isMinValue(IsSigned: false);
1816 default:
1817 return false;
1818 }
1819}
1820
1821bool AtomicExpandImpl::simplifyIdempotentRMW(AtomicRMWInst *RMWI) {
1822 if (auto ResultingLoad = TLI->lowerIdempotentRMWIntoFencedLoad(RMWI)) {
1823 tryExpandAtomicLoad(LI: ResultingLoad);
1824 return true;
1825 }
1826 return false;
1827}
1828
1829Value *AtomicExpandImpl::insertRMWCmpXchgLoop(
1830 IRBuilderBase &Builder, Type *ResultTy, Value *Addr, Align AddrAlign,
1831 AtomicOrdering MemOpOrder, SyncScope::ID SSID, bool IsVolatile,
1832 function_ref<Value *(IRBuilderBase &, Value *)> PerformOp,
1833 CreateCmpXchgInstFun CreateCmpXchg, Instruction *MetadataSrc) {
1834 LLVMContext &Ctx = Builder.getContext();
1835 BasicBlock *BB = Builder.GetInsertBlock();
1836 Function *F = BB->getParent();
1837
1838 // Given: atomicrmw some_op iN* %addr, iN %incr ordering
1839 //
1840 // The standard expansion we produce is:
1841 // [...]
1842 // %init_loaded = load atomic iN* %addr
1843 // br label %loop
1844 // loop:
1845 // %loaded = phi iN [ %init_loaded, %entry ], [ %new_loaded, %loop ]
1846 // %new = some_op iN %loaded, %incr
1847 // %pair = cmpxchg iN* %addr, iN %loaded, iN %new
1848 // %new_loaded = extractvalue { iN, i1 } %pair, 0
1849 // %success = extractvalue { iN, i1 } %pair, 1
1850 // br i1 %success, label %atomicrmw.end, label %loop
1851 // atomicrmw.end:
1852 // [...]
1853 BasicBlock *ExitBB =
1854 BB->splitBasicBlock(I: Builder.GetInsertPoint(), BBName: "atomicrmw.end");
1855 BasicBlock *LoopBB = BasicBlock::Create(Context&: Ctx, Name: "atomicrmw.start", Parent: F, InsertBefore: ExitBB);
1856
1857 // The split call above "helpfully" added a branch at the end of BB (to the
1858 // wrong place), but we want a load. It's easiest to just remove
1859 // the branch entirely.
1860 std::prev(x: BB->end())->eraseFromParent();
1861 Builder.SetInsertPoint(BB);
1862 LoadInst *InitLoaded = Builder.CreateAlignedLoad(Ty: ResultTy, Ptr: Addr, Align: AddrAlign);
1863 Builder.CreateBr(Dest: LoopBB);
1864
1865 // Start the main loop block now that we've taken care of the preliminaries.
1866 Builder.SetInsertPoint(LoopBB);
1867 PHINode *Loaded = Builder.CreatePHI(Ty: ResultTy, NumReservedValues: 2, Name: "loaded");
1868 Loaded->addIncoming(V: InitLoaded, BB);
1869
1870 // The initial load must be atomic with the same synchronization scope
1871 // to avoid a data race with concurrent stores. If the instruction being
1872 // emulated is volatile, issue a volatile load.
1873 // addIncoming is done first so that any replaceAllUsesWith calls during
1874 // normalization correctly update the PHI incoming value.
1875 InitLoaded->setVolatile(IsVolatile);
1876 if (TLI->shouldIssueAtomicLoadForAtomicEmulationLoop()) {
1877 InitLoaded->setAtomic(Ordering: AtomicOrdering::Monotonic, SSID);
1878 // The newly created load might need to be lowered further. Because it is
1879 // created in the same block as the atomicrmw, the AtomicExpand loop will
1880 // not process it again.
1881 processAtomicInstr(I: InitLoaded);
1882 }
1883
1884 Value *NewVal = PerformOp(Builder, Loaded);
1885
1886 Value *NewLoaded = nullptr;
1887 Value *Success = nullptr;
1888
1889 CreateCmpXchg(Builder, Addr, Loaded, NewVal, AddrAlign,
1890 MemOpOrder == AtomicOrdering::Unordered
1891 ? AtomicOrdering::Monotonic
1892 : MemOpOrder,
1893 SSID, IsVolatile, Success, NewLoaded, MetadataSrc);
1894 assert(Success && NewLoaded);
1895
1896 Loaded->addIncoming(V: NewLoaded, BB: LoopBB);
1897
1898 Instruction *CondBr = Builder.CreateCondBr(Cond: Success, True: ExitBB, False: LoopBB);
1899
1900 // Atomic RMW expands to a cmpxchg loop, Since precise branch weights
1901 // cannot be easily determined here, we mark the branch as "unknown" (50/50)
1902 // to prevent misleading optimizations.
1903 setExplicitlyUnknownBranchWeightsIfProfiled(I&: *CondBr, DEBUG_TYPE);
1904
1905 Builder.SetInsertPoint(ExitBB->begin());
1906 return NewLoaded;
1907}
1908
1909bool AtomicExpandImpl::tryExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) {
1910 unsigned MinCASSize = TLI->getMinCmpXchgSizeInBits() / 8;
1911 unsigned ValueSize = getAtomicOpSize(CASI: CI);
1912
1913 switch (TLI->shouldExpandAtomicCmpXchgInIR(AI: CI)) {
1914 default:
1915 llvm_unreachable("Unhandled case in tryExpandAtomicCmpXchg");
1916 case TargetLoweringBase::AtomicExpansionKind::None:
1917 if (ValueSize < MinCASSize)
1918 return expandPartwordCmpXchg(CI);
1919 return false;
1920 case TargetLoweringBase::AtomicExpansionKind::LLSC: {
1921 return expandAtomicCmpXchg(CI);
1922 }
1923 case TargetLoweringBase::AtomicExpansionKind::MaskedIntrinsic:
1924 expandAtomicCmpXchgToMaskedIntrinsic(CI);
1925 return true;
1926 case TargetLoweringBase::AtomicExpansionKind::NotAtomic:
1927 return lowerAtomicCmpXchgInst(CXI: CI);
1928 case TargetLoweringBase::AtomicExpansionKind::CustomExpand: {
1929 TLI->emitExpandAtomicCmpXchg(CI);
1930 return true;
1931 }
1932 }
1933}
1934
1935bool AtomicExpandImpl::expandAtomicRMWToCmpXchg(
1936 AtomicRMWInst *AI, CreateCmpXchgInstFun CreateCmpXchg) {
1937 ReplacementIRBuilder Builder(AI, AI->getDataLayout());
1938 Builder.setIsFPConstrained(
1939 AI->getFunction()->hasFnAttribute(Kind: Attribute::StrictFP));
1940
1941 // FIXME: If FP exceptions are observable, we should force them off for the
1942 // loop for the FP atomics.
1943 Value *Loaded = AtomicExpandImpl::insertRMWCmpXchgLoop(
1944 Builder, ResultTy: AI->getType(), Addr: AI->getPointerOperand(), AddrAlign: AI->getAlign(),
1945 MemOpOrder: AI->getOrdering(), SSID: AI->getSyncScopeID(), IsVolatile: AI->isVolatile(),
1946 PerformOp: [&](IRBuilderBase &Builder, Value *Loaded) {
1947 return buildAtomicRMWValue(Op: AI->getOperation(), Builder, Loaded,
1948 Val: AI->getValOperand());
1949 },
1950 CreateCmpXchg, /*MetadataSrc=*/AI);
1951
1952 AI->replaceAllUsesWith(V: Loaded);
1953 AI->eraseFromParent();
1954 return true;
1955}
1956
1957// In order to use one of the sized library calls such as
1958// __atomic_fetch_add_4, the alignment must be sufficient, the size
1959// must be one of the potentially-specialized sizes, and the value
1960// type must actually exist in C on the target (otherwise, the
1961// function wouldn't actually be defined.)
1962static bool canUseSizedAtomicCall(unsigned Size, Align Alignment,
1963 const DataLayout &DL) {
1964 // TODO: "LargestSize" is an approximation for "largest type that
1965 // you can express in C". It seems to be the case that int128 is
1966 // supported on all 64-bit platforms, otherwise only up to 64-bit
1967 // integers are supported. If we get this wrong, then we'll try to
1968 // call a sized libcall that doesn't actually exist. There should
1969 // really be some more reliable way in LLVM of determining integer
1970 // sizes which are valid in the target's C ABI...
1971 unsigned LargestSize = DL.getLargestLegalIntTypeSizeInBits() >= 64 ? 16 : 8;
1972 return Alignment >= Size &&
1973 (Size == 1 || Size == 2 || Size == 4 || Size == 8 || Size == 16) &&
1974 Size <= LargestSize;
1975}
1976
1977void AtomicExpandImpl::expandAtomicLoadToLibcall(LoadInst *I) {
1978 static const RTLIB::Libcall Libcalls[6] = {
1979 RTLIB::ATOMIC_LOAD, RTLIB::ATOMIC_LOAD_1, RTLIB::ATOMIC_LOAD_2,
1980 RTLIB::ATOMIC_LOAD_4, RTLIB::ATOMIC_LOAD_8, RTLIB::ATOMIC_LOAD_16};
1981 unsigned Size = getAtomicOpSize(LI: I);
1982
1983 bool Expanded = expandAtomicOpToLibcall(
1984 I, Size, Alignment: I->getAlign(), PointerOperand: I->getPointerOperand(), ValueOperand: nullptr, CASExpected: nullptr,
1985 Ordering: I->getOrdering(), Ordering2: AtomicOrdering::NotAtomic, Libcalls);
1986 if (!Expanded)
1987 handleUnsupportedAtomicSize(I, AtomicOpName: "atomic load");
1988}
1989
1990void AtomicExpandImpl::expandAtomicStoreToLibcall(StoreInst *I) {
1991 static const RTLIB::Libcall Libcalls[6] = {
1992 RTLIB::ATOMIC_STORE, RTLIB::ATOMIC_STORE_1, RTLIB::ATOMIC_STORE_2,
1993 RTLIB::ATOMIC_STORE_4, RTLIB::ATOMIC_STORE_8, RTLIB::ATOMIC_STORE_16};
1994 unsigned Size = getAtomicOpSize(SI: I);
1995
1996 bool Expanded = expandAtomicOpToLibcall(
1997 I, Size, Alignment: I->getAlign(), PointerOperand: I->getPointerOperand(), ValueOperand: I->getValueOperand(),
1998 CASExpected: nullptr, Ordering: I->getOrdering(), Ordering2: AtomicOrdering::NotAtomic, Libcalls);
1999 if (!Expanded)
2000 handleUnsupportedAtomicSize(I, AtomicOpName: "atomic store");
2001}
2002
2003void AtomicExpandImpl::expandAtomicCASToLibcall(AtomicCmpXchgInst *I,
2004 const Twine &AtomicOpName,
2005 Instruction *DiagnosticInst) {
2006 static const RTLIB::Libcall Libcalls[6] = {
2007 RTLIB::ATOMIC_COMPARE_EXCHANGE, RTLIB::ATOMIC_COMPARE_EXCHANGE_1,
2008 RTLIB::ATOMIC_COMPARE_EXCHANGE_2, RTLIB::ATOMIC_COMPARE_EXCHANGE_4,
2009 RTLIB::ATOMIC_COMPARE_EXCHANGE_8, RTLIB::ATOMIC_COMPARE_EXCHANGE_16};
2010 unsigned Size = getAtomicOpSize(CASI: I);
2011
2012 bool Expanded = expandAtomicOpToLibcall(
2013 I, Size, Alignment: I->getAlign(), PointerOperand: I->getPointerOperand(), ValueOperand: I->getNewValOperand(),
2014 CASExpected: I->getCompareOperand(), Ordering: I->getSuccessOrdering(), Ordering2: I->getFailureOrdering(),
2015 Libcalls);
2016 if (!Expanded)
2017 handleUnsupportedAtomicSize(I, AtomicOpName, DiagnosticInst);
2018}
2019
2020static ArrayRef<RTLIB::Libcall> GetRMWLibcall(AtomicRMWInst::BinOp Op) {
2021 static const RTLIB::Libcall LibcallsXchg[6] = {
2022 RTLIB::ATOMIC_EXCHANGE, RTLIB::ATOMIC_EXCHANGE_1,
2023 RTLIB::ATOMIC_EXCHANGE_2, RTLIB::ATOMIC_EXCHANGE_4,
2024 RTLIB::ATOMIC_EXCHANGE_8, RTLIB::ATOMIC_EXCHANGE_16};
2025 static const RTLIB::Libcall LibcallsAdd[6] = {
2026 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_ADD_1,
2027 RTLIB::ATOMIC_FETCH_ADD_2, RTLIB::ATOMIC_FETCH_ADD_4,
2028 RTLIB::ATOMIC_FETCH_ADD_8, RTLIB::ATOMIC_FETCH_ADD_16};
2029 static const RTLIB::Libcall LibcallsSub[6] = {
2030 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_SUB_1,
2031 RTLIB::ATOMIC_FETCH_SUB_2, RTLIB::ATOMIC_FETCH_SUB_4,
2032 RTLIB::ATOMIC_FETCH_SUB_8, RTLIB::ATOMIC_FETCH_SUB_16};
2033 static const RTLIB::Libcall LibcallsAnd[6] = {
2034 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_AND_1,
2035 RTLIB::ATOMIC_FETCH_AND_2, RTLIB::ATOMIC_FETCH_AND_4,
2036 RTLIB::ATOMIC_FETCH_AND_8, RTLIB::ATOMIC_FETCH_AND_16};
2037 static const RTLIB::Libcall LibcallsOr[6] = {
2038 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_OR_1,
2039 RTLIB::ATOMIC_FETCH_OR_2, RTLIB::ATOMIC_FETCH_OR_4,
2040 RTLIB::ATOMIC_FETCH_OR_8, RTLIB::ATOMIC_FETCH_OR_16};
2041 static const RTLIB::Libcall LibcallsXor[6] = {
2042 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_XOR_1,
2043 RTLIB::ATOMIC_FETCH_XOR_2, RTLIB::ATOMIC_FETCH_XOR_4,
2044 RTLIB::ATOMIC_FETCH_XOR_8, RTLIB::ATOMIC_FETCH_XOR_16};
2045 static const RTLIB::Libcall LibcallsNand[6] = {
2046 RTLIB::UNKNOWN_LIBCALL, RTLIB::ATOMIC_FETCH_NAND_1,
2047 RTLIB::ATOMIC_FETCH_NAND_2, RTLIB::ATOMIC_FETCH_NAND_4,
2048 RTLIB::ATOMIC_FETCH_NAND_8, RTLIB::ATOMIC_FETCH_NAND_16};
2049
2050 switch (Op) {
2051 case AtomicRMWInst::BAD_BINOP:
2052 llvm_unreachable("Should not have BAD_BINOP.");
2053 case AtomicRMWInst::Xchg:
2054 return ArrayRef(LibcallsXchg);
2055 case AtomicRMWInst::Add:
2056 return ArrayRef(LibcallsAdd);
2057 case AtomicRMWInst::Sub:
2058 return ArrayRef(LibcallsSub);
2059 case AtomicRMWInst::And:
2060 return ArrayRef(LibcallsAnd);
2061 case AtomicRMWInst::Or:
2062 return ArrayRef(LibcallsOr);
2063 case AtomicRMWInst::Xor:
2064 return ArrayRef(LibcallsXor);
2065 case AtomicRMWInst::Nand:
2066 return ArrayRef(LibcallsNand);
2067 case AtomicRMWInst::Max:
2068 case AtomicRMWInst::Min:
2069 case AtomicRMWInst::UMax:
2070 case AtomicRMWInst::UMin:
2071 case AtomicRMWInst::FMax:
2072 case AtomicRMWInst::FMin:
2073 case AtomicRMWInst::FMaximum:
2074 case AtomicRMWInst::FMinimum:
2075 case AtomicRMWInst::FMaximumNum:
2076 case AtomicRMWInst::FMinimumNum:
2077 case AtomicRMWInst::FAdd:
2078 case AtomicRMWInst::FSub:
2079 case AtomicRMWInst::UIncWrap:
2080 case AtomicRMWInst::UDecWrap:
2081 case AtomicRMWInst::USubCond:
2082 case AtomicRMWInst::USubSat:
2083 // No atomic libcalls are available for these.
2084 return {};
2085 }
2086 llvm_unreachable("Unexpected AtomicRMW operation.");
2087}
2088
2089void AtomicExpandImpl::expandAtomicRMWToLibcall(AtomicRMWInst *I) {
2090 ArrayRef<RTLIB::Libcall> Libcalls = GetRMWLibcall(Op: I->getOperation());
2091
2092 unsigned Size = getAtomicOpSize(RMWI: I);
2093
2094 bool Success = false;
2095 if (!Libcalls.empty())
2096 Success = expandAtomicOpToLibcall(
2097 I, Size, Alignment: I->getAlign(), PointerOperand: I->getPointerOperand(), ValueOperand: I->getValOperand(),
2098 CASExpected: nullptr, Ordering: I->getOrdering(), Ordering2: AtomicOrdering::NotAtomic, Libcalls);
2099
2100 // The expansion failed: either there were no libcalls at all for
2101 // the operation (min/max), or there were only size-specialized
2102 // libcalls (add/sub/etc) and we needed a generic. So, expand to a
2103 // CAS libcall, via a CAS loop, instead.
2104 if (!Success) {
2105 expandAtomicRMWToCmpXchg(
2106 AI: I, CreateCmpXchg: [this, I](IRBuilderBase &Builder, Value *Addr, Value *Loaded,
2107 Value *NewVal, Align Alignment, AtomicOrdering MemOpOrder,
2108 SyncScope::ID SSID, bool IsVolatile, Value *&Success,
2109 Value *&NewLoaded, Instruction *MetadataSrc) {
2110 // Create the CAS instruction normally...
2111 AtomicCmpXchgInst *Pair = Builder.CreateAtomicCmpXchg(
2112 Ptr: Addr, Cmp: Loaded, New: NewVal, Align: Alignment, SuccessOrdering: MemOpOrder,
2113 FailureOrdering: AtomicCmpXchgInst::getStrongestFailureOrdering(SuccessOrdering: MemOpOrder), SSID);
2114 Pair->setVolatile(IsVolatile);
2115 if (MetadataSrc)
2116 copyMetadataForAtomic(Dest&: *Pair, Source: *MetadataSrc);
2117
2118 Success = Builder.CreateExtractValue(Agg: Pair, Idxs: 1, Name: "success");
2119 NewLoaded = Builder.CreateExtractValue(Agg: Pair, Idxs: 0, Name: "newloaded");
2120
2121 // ...and then expand the CAS into a libcall.
2122 expandAtomicCASToLibcall(
2123 I: Pair,
2124 AtomicOpName: "atomicrmw " + AtomicRMWInst::getOperationName(Op: I->getOperation()),
2125 DiagnosticInst: MetadataSrc);
2126 });
2127 }
2128}
2129
2130// A helper routine for the above expandAtomic*ToLibcall functions.
2131//
2132// 'Libcalls' contains an array of enum values for the particular
2133// ATOMIC libcalls to be emitted. All of the other arguments besides
2134// 'I' are extracted from the Instruction subclass by the
2135// caller. Depending on the particular call, some will be null.
2136bool AtomicExpandImpl::expandAtomicOpToLibcall(
2137 Instruction *I, unsigned Size, Align Alignment, Value *PointerOperand,
2138 Value *ValueOperand, Value *CASExpected, AtomicOrdering Ordering,
2139 AtomicOrdering Ordering2, ArrayRef<RTLIB::Libcall> Libcalls) {
2140 assert(Libcalls.size() == 6);
2141
2142 LLVMContext &Ctx = I->getContext();
2143 Module *M = I->getModule();
2144 const DataLayout &DL = M->getDataLayout();
2145 IRBuilder<> Builder(I);
2146 IRBuilder<> AllocaBuilder(&I->getFunction()->getEntryBlock().front());
2147
2148 bool UseSizedLibcall = canUseSizedAtomicCall(Size, Alignment, DL);
2149 Type *SizedIntTy = Type::getIntNTy(C&: Ctx, N: Size * 8);
2150
2151 if (M->getTargetTriple().isOSWindows() && M->getTargetTriple().isX86_64() &&
2152 Size == 16) {
2153 // x86_64 Windows passes i128 as an XMM vector; on return, it is in
2154 // XMM0, and as a parameter, it is passed indirectly. The generic lowering
2155 // rules handles this correctly if we pass it as a v2i64 rather than
2156 // i128. This is what Clang does in the frontend for such types as well
2157 // (see WinX86_64ABIInfo::classify in Clang).
2158 SizedIntTy = FixedVectorType::get(ElementType: Type::getInt64Ty(C&: Ctx), NumElts: 2);
2159 }
2160
2161 const Align AllocaAlignment = DL.getPrefTypeAlign(Ty: SizedIntTy);
2162
2163 // TODO: the "order" argument type is "int", not int32. So
2164 // getInt32Ty may be wrong if the arch uses e.g. 16-bit ints.
2165 assert(Ordering != AtomicOrdering::NotAtomic && "expect atomic MO");
2166 Constant *OrderingVal =
2167 ConstantInt::get(Ty: Type::getInt32Ty(C&: Ctx), V: (int)toCABI(AO: Ordering));
2168 Constant *Ordering2Val = nullptr;
2169 if (CASExpected) {
2170 assert(Ordering2 != AtomicOrdering::NotAtomic && "expect atomic MO");
2171 Ordering2Val =
2172 ConstantInt::get(Ty: Type::getInt32Ty(C&: Ctx), V: (int)toCABI(AO: Ordering2));
2173 }
2174 bool HasResult = I->getType() != Type::getVoidTy(C&: Ctx);
2175
2176 RTLIB::Libcall RTLibType;
2177 if (UseSizedLibcall) {
2178 switch (Size) {
2179 case 1:
2180 RTLibType = Libcalls[1];
2181 break;
2182 case 2:
2183 RTLibType = Libcalls[2];
2184 break;
2185 case 4:
2186 RTLibType = Libcalls[3];
2187 break;
2188 case 8:
2189 RTLibType = Libcalls[4];
2190 break;
2191 case 16:
2192 RTLibType = Libcalls[5];
2193 break;
2194 }
2195 } else if (Libcalls[0] != RTLIB::UNKNOWN_LIBCALL) {
2196 RTLibType = Libcalls[0];
2197 } else {
2198 // Can't use sized function, and there's no generic for this
2199 // operation, so give up.
2200 return false;
2201 }
2202
2203 RTLIB::LibcallImpl LibcallImpl = LibcallLowering->getLibcallImpl(Call: RTLibType);
2204 if (LibcallImpl == RTLIB::Unsupported) {
2205 // This target does not implement the requested atomic libcall so give up.
2206 return false;
2207 }
2208
2209 // Build up the function call. There's two kinds. First, the sized
2210 // variants. These calls are going to be one of the following (with
2211 // N=1,2,4,8,16):
2212 // iN __atomic_load_N(iN *ptr, int ordering)
2213 // void __atomic_store_N(iN *ptr, iN val, int ordering)
2214 // iN __atomic_{exchange|fetch_*}_N(iN *ptr, iN val, int ordering)
2215 // bool __atomic_compare_exchange_N(iN *ptr, iN *expected, iN desired,
2216 // int success_order, int failure_order)
2217 //
2218 // Note that these functions can be used for non-integer atomic
2219 // operations, the values just need to be bitcast to integers on the
2220 // way in and out.
2221 //
2222 // And, then, the generic variants. They look like the following:
2223 // void __atomic_load(size_t size, void *ptr, void *ret, int ordering)
2224 // void __atomic_store(size_t size, void *ptr, void *val, int ordering)
2225 // void __atomic_exchange(size_t size, void *ptr, void *val, void *ret,
2226 // int ordering)
2227 // bool __atomic_compare_exchange(size_t size, void *ptr, void *expected,
2228 // void *desired, int success_order,
2229 // int failure_order)
2230 //
2231 // The different signatures are built up depending on the
2232 // 'UseSizedLibcall', 'CASExpected', 'ValueOperand', and 'HasResult'
2233 // variables.
2234
2235 AllocaInst *AllocaCASExpected = nullptr;
2236 AllocaInst *AllocaValue = nullptr;
2237 AllocaInst *AllocaResult = nullptr;
2238
2239 Type *ResultTy;
2240 SmallVector<Value *, 6> Args;
2241 AttributeList Attr;
2242
2243 // 'size' argument.
2244 if (!UseSizedLibcall) {
2245 // Note, getIntPtrType is assumed equivalent to size_t.
2246 Args.push_back(Elt: ConstantInt::get(Ty: DL.getIntPtrType(C&: Ctx), V: Size));
2247 }
2248
2249 // 'ptr' argument.
2250 // note: This assumes all address spaces share a common libfunc
2251 // implementation and that addresses are convertable. For systems without
2252 // that property, we'd need to extend this mechanism to support AS-specific
2253 // families of atomic intrinsics.
2254 Value *PtrVal = PointerOperand;
2255 PtrVal = Builder.CreateAddrSpaceCast(V: PtrVal, DestTy: PointerType::getUnqual(C&: Ctx));
2256 Args.push_back(Elt: PtrVal);
2257
2258 // 'expected' argument, if present.
2259 if (CASExpected) {
2260 AllocaCASExpected = AllocaBuilder.CreateAlloca(Ty: CASExpected->getType());
2261 AllocaCASExpected->setAlignment(AllocaAlignment);
2262 Builder.CreateLifetimeStart(Ptr: AllocaCASExpected);
2263 Builder.CreateAlignedStore(Val: CASExpected, Ptr: AllocaCASExpected, Align: AllocaAlignment);
2264 Args.push_back(Elt: AllocaCASExpected);
2265 }
2266
2267 // 'val' argument ('desired' for cas), if present.
2268 if (ValueOperand) {
2269 if (UseSizedLibcall) {
2270 Value *IntValue =
2271 Builder.CreateBitPreservingCastChain(DL, V: ValueOperand, NewTy: SizedIntTy);
2272 Args.push_back(Elt: IntValue);
2273 } else {
2274 AllocaValue = AllocaBuilder.CreateAlloca(Ty: ValueOperand->getType());
2275 AllocaValue->setAlignment(AllocaAlignment);
2276 Builder.CreateLifetimeStart(Ptr: AllocaValue);
2277 Builder.CreateAlignedStore(Val: ValueOperand, Ptr: AllocaValue, Align: AllocaAlignment);
2278 Args.push_back(Elt: AllocaValue);
2279 }
2280 }
2281
2282 // 'ret' argument.
2283 if (!CASExpected && HasResult && !UseSizedLibcall) {
2284 AllocaResult = AllocaBuilder.CreateAlloca(Ty: I->getType());
2285 AllocaResult->setAlignment(AllocaAlignment);
2286 Builder.CreateLifetimeStart(Ptr: AllocaResult);
2287 Args.push_back(Elt: AllocaResult);
2288 }
2289
2290 // 'ordering' ('success_order' for cas) argument.
2291 Args.push_back(Elt: OrderingVal);
2292
2293 // 'failure_order' argument, if present.
2294 if (Ordering2Val)
2295 Args.push_back(Elt: Ordering2Val);
2296
2297 // Now, the return type.
2298 if (CASExpected) {
2299 ResultTy = Type::getInt1Ty(C&: Ctx);
2300 Attr = Attr.addRetAttribute(C&: Ctx, Kind: Attribute::ZExt);
2301 } else if (HasResult && UseSizedLibcall)
2302 ResultTy = SizedIntTy;
2303 else
2304 ResultTy = Type::getVoidTy(C&: Ctx);
2305
2306 // Done with setting up arguments and return types, create the call:
2307 SmallVector<Type *, 6> ArgTys;
2308 for (Value *Arg : Args)
2309 ArgTys.push_back(Elt: Arg->getType());
2310 FunctionType *FnType = FunctionType::get(Result: ResultTy, Params: ArgTys, isVarArg: false);
2311 FunctionCallee LibcallFn = M->getOrInsertFunction(
2312 Name: RTLIB::RuntimeLibcallsInfo::getLibcallImplName(CallImpl: LibcallImpl), T: FnType,
2313 AttributeList: Attr);
2314 CallInst *Call = Builder.CreateCall(Callee: LibcallFn, Args);
2315 Call->setAttributes(Attr);
2316 Value *Result = Call;
2317
2318 // And then, extract the results...
2319 if (ValueOperand && !UseSizedLibcall)
2320 Builder.CreateLifetimeEnd(Ptr: AllocaValue);
2321
2322 if (CASExpected) {
2323 // The final result from the CAS is {load of 'expected' alloca, bool result
2324 // from call}
2325 Type *FinalResultTy = I->getType();
2326 Value *V = PoisonValue::get(T: FinalResultTy);
2327 Value *ExpectedOut = Builder.CreateAlignedLoad(
2328 Ty: CASExpected->getType(), Ptr: AllocaCASExpected, Align: AllocaAlignment);
2329 Builder.CreateLifetimeEnd(Ptr: AllocaCASExpected);
2330 V = Builder.CreateInsertValue(Agg: V, Val: ExpectedOut, Idxs: 0);
2331 V = Builder.CreateInsertValue(Agg: V, Val: Result, Idxs: 1);
2332 I->replaceAllUsesWith(V);
2333 } else if (HasResult) {
2334 Value *V;
2335 if (UseSizedLibcall) {
2336 // Add bitcasts from Result's scalar type to I's <n x ptr> vector type
2337 auto *PtrTy = dyn_cast<PointerType>(Val: I->getType()->getScalarType());
2338 auto *VTy = dyn_cast<VectorType>(Val: I->getType());
2339 if (VTy && PtrTy && !Result->getType()->isVectorTy()) {
2340 unsigned AS = PtrTy->getAddressSpace();
2341 Value *BC = Builder.CreateBitCast(
2342 V: Result, DestTy: VTy->getWithNewType(EltTy: DL.getIntPtrType(C&: Ctx, AddressSpace: AS)));
2343 V = Builder.CreateIntToPtr(V: BC, DestTy: I->getType());
2344 } else
2345 V = Builder.CreateBitOrPointerCast(V: Result, DestTy: I->getType());
2346 } else {
2347 V = Builder.CreateAlignedLoad(Ty: I->getType(), Ptr: AllocaResult,
2348 Align: AllocaAlignment);
2349 Builder.CreateLifetimeEnd(Ptr: AllocaResult);
2350 }
2351 I->replaceAllUsesWith(V);
2352 }
2353 I->eraseFromParent();
2354 return true;
2355}
2356