| 1 | //===- DXILShaderFlags.cpp - DXIL Shader Flags helper objects -------------===// |
| 2 | // |
| 3 | // Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions. |
| 4 | // See https://llvm.org/LICENSE.txt for license information. |
| 5 | // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception |
| 6 | // |
| 7 | //===----------------------------------------------------------------------===// |
| 8 | /// |
| 9 | /// \file This file contains helper objects and APIs for working with DXIL |
| 10 | /// Shader Flags. |
| 11 | /// |
| 12 | //===----------------------------------------------------------------------===// |
| 13 | |
| 14 | #include "DXILShaderFlags.h" |
| 15 | #include "DirectX.h" |
| 16 | #include "llvm/ADT/SCCIterator.h" |
| 17 | #include "llvm/ADT/SmallVector.h" |
| 18 | #include "llvm/Analysis/CallGraph.h" |
| 19 | #include "llvm/Analysis/DXILResource.h" |
| 20 | #include "llvm/IR/Attributes.h" |
| 21 | #include "llvm/IR/DiagnosticInfo.h" |
| 22 | #include "llvm/IR/Instruction.h" |
| 23 | #include "llvm/IR/Instructions.h" |
| 24 | #include "llvm/IR/IntrinsicInst.h" |
| 25 | #include "llvm/IR/Intrinsics.h" |
| 26 | #include "llvm/IR/IntrinsicsDirectX.h" |
| 27 | #include "llvm/IR/Module.h" |
| 28 | #include "llvm/InitializePasses.h" |
| 29 | #include "llvm/Support/FormatVariadic.h" |
| 30 | #include "llvm/Support/raw_ostream.h" |
| 31 | |
| 32 | using namespace llvm; |
| 33 | using namespace llvm::dxil; |
| 34 | |
| 35 | static bool hasUAVsAtEveryStage(const DXILResourceMap &DRM, |
| 36 | const ModuleMetadataInfo &MMDI) { |
| 37 | // Heap resources do not count towards hasUAVsAtEveryStage. |
| 38 | bool HasUAVWithBinding = any_of( |
| 39 | Range: DRM.uavs(), P: [](const ResourceInfo &RI) { return RI.hasBinding(); }); |
| 40 | if (!HasUAVWithBinding) |
| 41 | return false; |
| 42 | |
| 43 | switch (MMDI.ShaderProfile) { |
| 44 | default: |
| 45 | return false; |
| 46 | case Triple::EnvironmentType::Compute: |
| 47 | case Triple::EnvironmentType::Pixel: |
| 48 | return false; |
| 49 | case Triple::EnvironmentType::Vertex: |
| 50 | case Triple::EnvironmentType::Geometry: |
| 51 | case Triple::EnvironmentType::Hull: |
| 52 | case Triple::EnvironmentType::Domain: |
| 53 | return true; |
| 54 | case Triple::EnvironmentType::Library: |
| 55 | case Triple::EnvironmentType::RayGeneration: |
| 56 | case Triple::EnvironmentType::Intersection: |
| 57 | case Triple::EnvironmentType::AnyHit: |
| 58 | case Triple::EnvironmentType::ClosestHit: |
| 59 | case Triple::EnvironmentType::Miss: |
| 60 | case Triple::EnvironmentType::Callable: |
| 61 | case Triple::EnvironmentType::Mesh: |
| 62 | case Triple::EnvironmentType::Amplification: |
| 63 | return MMDI.ValidatorVersion < VersionTuple(1, 8); |
| 64 | } |
| 65 | } |
| 66 | |
| 67 | static bool checkWaveOps(Intrinsic::ID IID) { |
| 68 | // Currently unsupported intrinsics |
| 69 | // case Intrinsic::dx_wave_readfirst: |
| 70 | // case Intrinsic::dx_wave_reduce.and: |
| 71 | // case Intrinsic::dx_wave_reduce.or: |
| 72 | // case Intrinsic::dx_wave_reduce.xor: |
| 73 | // case Intrinsic::dx_wave_prefixop: |
| 74 | // case Intrinsic::dx_quad.readat: |
| 75 | // case Intrinsic::dx_quad.readacrossy: |
| 76 | // case Intrinsic::dx_quad.readacrossdiagonal: |
| 77 | // case Intrinsic::dx_wave_prefixballot: |
| 78 | // case Intrinsic::dx_wave_match: |
| 79 | // case Intrinsic::dx_wavemulti.*: |
| 80 | // case Intrinsic::dx_wavemulti.ballot: |
| 81 | // case Intrinsic::dx_quad.vote: |
| 82 | switch (IID) { |
| 83 | default: |
| 84 | return false; |
| 85 | case Intrinsic::dx_wave_is_first_lane: |
| 86 | case Intrinsic::dx_wave_getlaneindex: |
| 87 | case Intrinsic::dx_wave_get_lane_count: |
| 88 | case Intrinsic::dx_wave_any: |
| 89 | case Intrinsic::dx_wave_all_equal: |
| 90 | case Intrinsic::dx_wave_all: |
| 91 | case Intrinsic::dx_wave_readlane: |
| 92 | case Intrinsic::dx_wave_readlane_first: |
| 93 | case Intrinsic::dx_wave_active_countbits: |
| 94 | case Intrinsic::dx_wave_ballot: |
| 95 | case Intrinsic::dx_wave_prefix_bit_count: |
| 96 | // Wave Active Op Variants |
| 97 | case Intrinsic::dx_wave_reduce_or: |
| 98 | case Intrinsic::dx_wave_reduce_xor: |
| 99 | case Intrinsic::dx_wave_reduce_and: |
| 100 | case Intrinsic::dx_wave_reduce_sum: |
| 101 | case Intrinsic::dx_wave_reduce_usum: |
| 102 | case Intrinsic::dx_wave_product: |
| 103 | case Intrinsic::dx_wave_uproduct: |
| 104 | case Intrinsic::dx_wave_reduce_max: |
| 105 | case Intrinsic::dx_wave_reduce_umax: |
| 106 | case Intrinsic::dx_wave_reduce_min: |
| 107 | case Intrinsic::dx_wave_reduce_umin: |
| 108 | // Wave Prefix Op Variants |
| 109 | case Intrinsic::dx_wave_prefix_sum: |
| 110 | case Intrinsic::dx_wave_prefix_usum: |
| 111 | case Intrinsic::dx_wave_prefix_product: |
| 112 | case Intrinsic::dx_wave_prefix_uproduct: |
| 113 | // Quad Op Variants |
| 114 | case Intrinsic::dx_quad_read_across_x: |
| 115 | case Intrinsic::dx_quad_read_across_y: |
| 116 | case Intrinsic::dx_quad_read_across_diagonal: |
| 117 | return true; |
| 118 | } |
| 119 | } |
| 120 | |
| 121 | static bool checkDoubleExtensionOps(Intrinsic::ID IID) { |
| 122 | switch (IID) { |
| 123 | default: |
| 124 | return false; |
| 125 | case Intrinsic::fma: |
| 126 | return true; |
| 127 | } |
| 128 | } |
| 129 | |
| 130 | /// Texture load and sample operations accept "programmable offsets", i.e. |
| 131 | /// offsets that are not compile-time constants. Such offsets require the |
| 132 | /// AdvancedTextureOps shader feature flag. Returns true if \p II is one of |
| 133 | /// those operations and its offsets operand is not a constant. |
| 134 | static bool checkAdvancedTextureOps(const IntrinsicInst &II) { |
| 135 | // TODO: (#116137) Several other DXIL ops also require this feature flag, but |
| 136 | // none of them can be generated yet: |
| 137 | // - SampleCmp, SampleCmpBias, SampleCmpGrad and SampleCmpLevelZero set the |
| 138 | // flag for non-constant offsets, exactly like the ops handled below. |
| 139 | // - SampleCmpLevel, TextureGatherRaw and TextureStoreSample set the flag |
| 140 | // unconditionally, and have no intrinsics yet. |
| 141 | |
| 142 | // The offsets operand index differs between the intrinsics. |
| 143 | unsigned OffsetsIdx; |
| 144 | switch (II.getIntrinsicID()) { |
| 145 | default: |
| 146 | return false; |
| 147 | case Intrinsic::dx_resource_load_level: |
| 148 | case Intrinsic::dx_resource_sample: |
| 149 | case Intrinsic::dx_resource_sample_clamp: |
| 150 | OffsetsIdx = 3; |
| 151 | break; |
| 152 | case Intrinsic::dx_resource_samplebias: |
| 153 | case Intrinsic::dx_resource_samplebias_clamp: |
| 154 | case Intrinsic::dx_resource_samplelevel: |
| 155 | OffsetsIdx = 4; |
| 156 | break; |
| 157 | case Intrinsic::dx_resource_samplegrad: |
| 158 | case Intrinsic::dx_resource_samplegrad_clamp: |
| 159 | OffsetsIdx = 5; |
| 160 | break; |
| 161 | } |
| 162 | return !isa<Constant>(Val: II.getArgOperand(i: OffsetsIdx)); |
| 163 | } |
| 164 | |
| 165 | static bool isOptimizationDisabled(const Module &M) { |
| 166 | const StringRef Key = "dx.disable_optimizations" ; |
| 167 | if (auto *Flag = mdconst::extract_or_null<ConstantInt>(MD: M.getModuleFlag(Key))) |
| 168 | return Flag->getValue().getBoolValue(); |
| 169 | return false; |
| 170 | } |
| 171 | |
| 172 | // Checks to see if the status bit from a load with status |
| 173 | // instruction is ever extracted. If it is, the module needs |
| 174 | // to have the TiledResources shader flag set. |
| 175 | bool (const IntrinsicInst &II) { |
| 176 | [[maybe_unused]] Intrinsic::ID IID = II.getIntrinsicID(); |
| 177 | assert(IID == Intrinsic::dx_resource_load_typedbuffer || |
| 178 | IID == Intrinsic::dx_resource_load_rawbuffer && |
| 179 | "unexpected intrinsic ID" ); |
| 180 | for (const User *U : II.users()) { |
| 181 | if (const ExtractValueInst *EVI = dyn_cast<ExtractValueInst>(Val: U)) { |
| 182 | // Resource load operations return a {result, status} pair. |
| 183 | // Check if we extract the status |
| 184 | if (EVI->getNumIndices() == 1 && EVI->getIndices()[0] == 1) |
| 185 | return true; |
| 186 | } |
| 187 | } |
| 188 | |
| 189 | return false; |
| 190 | } |
| 191 | |
| 192 | /// Update the shader flags mask based on the given instruction. |
| 193 | /// \param CSF Shader flags mask to update. |
| 194 | /// \param I Instruction to check. |
| 195 | void ModuleShaderFlags::updateFunctionFlags(ComputedShaderFlags &CSF, |
| 196 | const Instruction &I, |
| 197 | DXILResourceTypeMap &DRTM, |
| 198 | const ModuleMetadataInfo &MMDI) { |
| 199 | if (!CSF.Doubles) |
| 200 | CSF.Doubles = I.getType()->getScalarType()->isDoubleTy(); |
| 201 | |
| 202 | if (!CSF.Doubles) { |
| 203 | for (const Value *Op : I.operands()) { |
| 204 | if (Op->getType()->getScalarType()->isDoubleTy()) { |
| 205 | CSF.Doubles = true; |
| 206 | break; |
| 207 | } |
| 208 | } |
| 209 | } |
| 210 | |
| 211 | if (CSF.Doubles) { |
| 212 | switch (I.getOpcode()) { |
| 213 | case Instruction::FDiv: |
| 214 | case Instruction::UIToFP: |
| 215 | case Instruction::SIToFP: |
| 216 | case Instruction::FPToUI: |
| 217 | case Instruction::FPToSI: |
| 218 | CSF.DX11_1_DoubleExtensions = true; |
| 219 | break; |
| 220 | } |
| 221 | } |
| 222 | |
| 223 | if (!CSF.LowPrecisionPresent) |
| 224 | CSF.LowPrecisionPresent = I.getType()->getScalarType()->isIntegerTy(BitWidth: 16) || |
| 225 | I.getType()->getScalarType()->isHalfTy(); |
| 226 | |
| 227 | if (!CSF.LowPrecisionPresent) { |
| 228 | for (const Value *Op : I.operands()) { |
| 229 | if (Op->getType()->getScalarType()->isIntegerTy(BitWidth: 16) || |
| 230 | Op->getType()->getScalarType()->isHalfTy()) { |
| 231 | CSF.LowPrecisionPresent = true; |
| 232 | break; |
| 233 | } |
| 234 | } |
| 235 | } |
| 236 | |
| 237 | if (CSF.LowPrecisionPresent) { |
| 238 | if (CSF.NativeLowPrecisionMode) |
| 239 | CSF.NativeLowPrecision = true; |
| 240 | else |
| 241 | CSF.MinimumPrecision = true; |
| 242 | } |
| 243 | |
| 244 | if (!CSF.Int64Ops) |
| 245 | CSF.Int64Ops = I.getType()->getScalarType()->isIntegerTy(BitWidth: 64); |
| 246 | |
| 247 | if (!CSF.Int64Ops && !isa<LifetimeIntrinsic>(Val: &I)) { |
| 248 | for (const Value *Op : I.operands()) { |
| 249 | if (Op->getType()->getScalarType()->isIntegerTy(BitWidth: 64)) { |
| 250 | CSF.Int64Ops = true; |
| 251 | break; |
| 252 | } |
| 253 | } |
| 254 | } |
| 255 | |
| 256 | if (const auto *II = dyn_cast<IntrinsicInst>(Val: &I)) { |
| 257 | CSF.AdvancedTextureOps |= checkAdvancedTextureOps(II: *II); |
| 258 | |
| 259 | switch (II->getIntrinsicID()) { |
| 260 | default: |
| 261 | break; |
| 262 | case Intrinsic::dx_resource_handlefrombinding: { |
| 263 | dxil::ResourceTypeInfo &RTI = DRTM[cast<TargetExtType>(Val: II->getType())]; |
| 264 | |
| 265 | // Set ResMayNotAlias if DXIL validator version >= 1.8 and the function |
| 266 | // uses UAVs |
| 267 | if (!CSF.ResMayNotAlias && CanSetResMayNotAlias && |
| 268 | MMDI.ValidatorVersion >= VersionTuple(1, 8) && RTI.isUAV()) |
| 269 | CSF.ResMayNotAlias = true; |
| 270 | |
| 271 | switch (RTI.getResourceKind()) { |
| 272 | case dxil::ResourceKind::StructuredBuffer: |
| 273 | case dxil::ResourceKind::RawBuffer: |
| 274 | CSF.EnableRawAndStructuredBuffers = true; |
| 275 | break; |
| 276 | default: |
| 277 | break; |
| 278 | } |
| 279 | break; |
| 280 | } |
| 281 | case Intrinsic::dx_resource_handlefromheap: { |
| 282 | dxil::ResourceTypeInfo &RTI = DRTM[cast<TargetExtType>(Val: II->getType())]; |
| 283 | bool IsSamplerHeap = RTI.isSampler(); |
| 284 | CSF.SamplerDescriptorHeapIndexing |= IsSamplerHeap; |
| 285 | CSF.ResourceDescriptorHeapIndexing |= !IsSamplerHeap; |
| 286 | |
| 287 | if (!CSF.ResMayNotAlias && CanSetResMayNotAlias && RTI.isUAV() && |
| 288 | MMDI.ValidatorVersion >= VersionTuple(1, 8)) { |
| 289 | CSF.ResMayNotAlias = true; |
| 290 | } |
| 291 | break; |
| 292 | } |
| 293 | case Intrinsic::dx_resource_load_level: |
| 294 | case Intrinsic::dx_resource_load_typedbuffer: { |
| 295 | dxil::ResourceTypeInfo &RTI = |
| 296 | DRTM[cast<TargetExtType>(Val: II->getArgOperand(i: 0)->getType())]; |
| 297 | if (RTI.isTyped() && RTI.isUAV()) |
| 298 | CSF.TypedUAVLoadAdditionalFormats |= RTI.getTyped().ElementCount > 1; |
| 299 | if (II->getIntrinsicID() == Intrinsic::dx_resource_load_typedbuffer && |
| 300 | !CSF.TiledResources && checkIfStatusIsExtracted(II: *II)) |
| 301 | CSF.TiledResources = true; |
| 302 | break; |
| 303 | } |
| 304 | case Intrinsic::dx_resource_load_rawbuffer: { |
| 305 | if (!CSF.TiledResources && checkIfStatusIsExtracted(II: *II)) |
| 306 | CSF.TiledResources = true; |
| 307 | break; |
| 308 | } |
| 309 | case Intrinsic::dx_resource_atomic_binop: |
| 310 | case Intrinsic::dx_resource_atomic_compare_exchange: { |
| 311 | if (II->getType()->isIntegerTy(BitWidth: 64)) { |
| 312 | dxil::ResourceTypeInfo &RTI = |
| 313 | DRTM[cast<TargetExtType>(Val: II->getArgOperand(i: 0)->getType())]; |
| 314 | if (RTI.isTyped()) |
| 315 | CSF.AtomicInt64OnTypedResource = true; |
| 316 | // TODO(https://github.com/llvm/llvm-project/issues/116152): Set |
| 317 | // AtomicInt64OnHeapResource when heap-resource intrinsics are added. |
| 318 | } |
| 319 | break; |
| 320 | } |
| 321 | } |
| 322 | } |
| 323 | // 64-bit atomics on groupshared memory (address space 3). |
| 324 | if (const auto *ARMW = dyn_cast<AtomicRMWInst>(Val: &I)) { |
| 325 | if (ARMW->getValOperand()->getType()->isIntegerTy(BitWidth: 64) && |
| 326 | ARMW->getPointerAddressSpace() == 3) |
| 327 | CSF.AtomicInt64OnGroupShared = true; |
| 328 | } else if (const auto *AXCG = dyn_cast<AtomicCmpXchgInst>(Val: &I)) { |
| 329 | if (AXCG->getNewValOperand()->getType()->isIntegerTy(BitWidth: 64) && |
| 330 | AXCG->getPointerAddressSpace() == 3) |
| 331 | CSF.AtomicInt64OnGroupShared = true; |
| 332 | } |
| 333 | // Handle call instructions |
| 334 | if (auto *CI = dyn_cast<CallInst>(Val: &I)) { |
| 335 | const Function *CF = CI->getCalledFunction(); |
| 336 | // Merge-in shader flags mask of the called function in the current module |
| 337 | if (FunctionFlags.contains(Val: CF)) |
| 338 | CSF.merge(CSF: FunctionFlags[CF]); |
| 339 | |
| 340 | CSF.DX11_1_DoubleExtensions |= |
| 341 | checkDoubleExtensionOps(IID: CI->getIntrinsicID()); |
| 342 | CSF.WaveOps |= checkWaveOps(IID: CI->getIntrinsicID()); |
| 343 | } |
| 344 | } |
| 345 | |
| 346 | /// Set shader flags that apply to all functions within the module |
| 347 | ComputedShaderFlags |
| 348 | ModuleShaderFlags::gatherGlobalModuleFlags(const Module &M, |
| 349 | const DXILResourceMap &DRM, |
| 350 | const ModuleMetadataInfo &MMDI) { |
| 351 | |
| 352 | ComputedShaderFlags CSF; |
| 353 | |
| 354 | CSF.DisableOptimizations = isOptimizationDisabled(M); |
| 355 | |
| 356 | CSF.UAVsAtEveryStage = hasUAVsAtEveryStage(DRM, MMDI); |
| 357 | |
| 358 | // Set the Max64UAVs flag if the number of UAVs is > 8 |
| 359 | uint32_t NumUAVs = 0; |
| 360 | for (auto &UAV : DRM.uavs()) { |
| 361 | // Heap resources do not count towards Max64UAVs flag. |
| 362 | if (!UAV.hasBinding()) |
| 363 | continue; |
| 364 | if (MMDI.ValidatorVersion < VersionTuple(1, 6)) { |
| 365 | NumUAVs++; |
| 366 | } else { // MMDI.ValidatorVersion >= VersionTuple(1, 6) |
| 367 | uint32_t Size = UAV.getSize(); |
| 368 | uint32_t NewNum = NumUAVs + (Size == 0 ? ~0U : Size); |
| 369 | if (NewNum < NumUAVs) |
| 370 | NewNum = ~0U; |
| 371 | NumUAVs = NewNum; |
| 372 | } |
| 373 | } |
| 374 | if (NumUAVs > 8) |
| 375 | CSF.Max64UAVs = true; |
| 376 | |
| 377 | // Set the module flag that enables native low-precision execution mode. |
| 378 | // NativeLowPrecisionMode can only be set when the command line option |
| 379 | // -enable-16bit-types is provided. This is indicated by the dx.nativelowprec |
| 380 | // module flag being set |
| 381 | // This flag is needed even if the module does not use 16-bit types because a |
| 382 | // corresponding debug module may include 16-bit types, and tools that use the |
| 383 | // debug module may expect it to have the same flags as the original |
| 384 | if (auto *NativeLowPrec = mdconst::extract_or_null<ConstantInt>( |
| 385 | MD: M.getModuleFlag(Key: "dx.nativelowprec" ))) |
| 386 | if (MMDI.ShaderModelVersion >= VersionTuple(6, 2)) |
| 387 | CSF.NativeLowPrecisionMode = NativeLowPrec->getValue().getBoolValue(); |
| 388 | |
| 389 | // Set ResMayNotAlias to true if DXIL validator version < 1.8 and there |
| 390 | // are UAVs present globally. |
| 391 | if (CanSetResMayNotAlias && MMDI.ValidatorVersion < VersionTuple(1, 8)) |
| 392 | CSF.ResMayNotAlias = !DRM.uavs().empty(); |
| 393 | |
| 394 | // The command line option -all-resources-bound will set the |
| 395 | // dx.allresourcesbound module flag to 1 |
| 396 | if (auto *AllResourcesBound = mdconst::extract_or_null<ConstantInt>( |
| 397 | MD: M.getModuleFlag(Key: "dx.allresourcesbound" ))) |
| 398 | if (AllResourcesBound->getValue().getBoolValue()) |
| 399 | CSF.AllResourcesBound = true; |
| 400 | |
| 401 | return CSF; |
| 402 | } |
| 403 | |
| 404 | /// Construct ModuleShaderFlags for module Module M |
| 405 | void ModuleShaderFlags::initialize(Module &M, DXILResourceTypeMap &DRTM, |
| 406 | const DXILResourceMap &DRM, |
| 407 | const ModuleMetadataInfo &MMDI) { |
| 408 | |
| 409 | CanSetResMayNotAlias = MMDI.DXILVersion >= VersionTuple(1, 7); |
| 410 | // The command line option -res-may-alias will set the dx.resmayalias module |
| 411 | // flag to 1, thereby disabling the ability to set the ResMayNotAlias flag |
| 412 | if (auto *ResMayAlias = mdconst::extract_or_null<ConstantInt>( |
| 413 | MD: M.getModuleFlag(Key: "dx.resmayalias" ))) |
| 414 | if (ResMayAlias->getValue().getBoolValue()) |
| 415 | CanSetResMayNotAlias = false; |
| 416 | |
| 417 | ComputedShaderFlags GlobalSFMask = gatherGlobalModuleFlags(M, DRM, MMDI); |
| 418 | |
| 419 | CallGraph CG(M); |
| 420 | |
| 421 | // Compute Shader Flags Mask for all functions using post-order visit of SCC |
| 422 | // of the call graph. |
| 423 | for (scc_iterator<CallGraph *> SCCI = scc_begin(G: &CG); !SCCI.isAtEnd(); |
| 424 | ++SCCI) { |
| 425 | const std::vector<CallGraphNode *> &CurSCC = *SCCI; |
| 426 | |
| 427 | // Union of shader masks of all functions in CurSCC |
| 428 | ComputedShaderFlags SCCSF; |
| 429 | // List of functions in CurSCC that are neither external nor declarations |
| 430 | // and hence whose flags are collected |
| 431 | SmallVector<Function *> CurSCCFuncs; |
| 432 | for (CallGraphNode *CGN : CurSCC) { |
| 433 | Function *F = CGN->getFunction(); |
| 434 | if (!F) |
| 435 | continue; |
| 436 | |
| 437 | if (F->isDeclaration()) { |
| 438 | assert(!F->getName().starts_with("dx.op." ) && |
| 439 | "DXIL Shader Flag analysis should not be run post-lowering." ); |
| 440 | continue; |
| 441 | } |
| 442 | |
| 443 | ComputedShaderFlags CSF = GlobalSFMask; |
| 444 | for (const auto &BB : *F) |
| 445 | for (const auto &I : BB) |
| 446 | updateFunctionFlags(CSF, I, DRTM, MMDI); |
| 447 | // Update combined shader flags mask for all functions in this SCC |
| 448 | SCCSF.merge(CSF); |
| 449 | |
| 450 | CurSCCFuncs.push_back(Elt: F); |
| 451 | } |
| 452 | |
| 453 | // Update combined shader flags mask for all functions of the module |
| 454 | CombinedSFMask.merge(CSF: SCCSF); |
| 455 | |
| 456 | // Shader flags mask of each of the functions in an SCC of the call graph is |
| 457 | // the union of all functions in the SCC. Update shader flags masks of |
| 458 | // functions in CurSCC accordingly. This is trivially true if SCC contains |
| 459 | // one function. |
| 460 | for (Function *F : CurSCCFuncs) |
| 461 | // Merge SCCSF with that of F |
| 462 | FunctionFlags[F].merge(CSF: SCCSF); |
| 463 | } |
| 464 | } |
| 465 | |
| 466 | void ComputedShaderFlags::print(raw_ostream &OS) const { |
| 467 | uint64_t FlagVal = (uint64_t) * this; |
| 468 | OS << formatv(Fmt: "; Shader Flags Value: {0:x8}\n;\n" , Vals&: FlagVal); |
| 469 | if (FlagVal == 0) |
| 470 | return; |
| 471 | OS << "; Note: shader requires additional functionality:\n" ; |
| 472 | #define SHADER_FEATURE_FLAG(FeatureBit, DxilModuleNum, FlagName, Str) \ |
| 473 | if (FlagName) \ |
| 474 | (OS << ";").indent(7) << Str << "\n"; |
| 475 | #include "llvm/BinaryFormat/DXContainerConstants.def" |
| 476 | OS << "; Note: extra DXIL module flags:\n" ; |
| 477 | #define DXIL_MODULE_FLAG(DxilModuleBit, FlagName, Str) \ |
| 478 | if (FlagName) \ |
| 479 | (OS << ";").indent(7) << Str << "\n"; |
| 480 | #include "llvm/BinaryFormat/DXContainerConstants.def" |
| 481 | OS << ";\n" ; |
| 482 | } |
| 483 | |
| 484 | /// Return the shader flags mask of the specified function Func. |
| 485 | const ComputedShaderFlags & |
| 486 | ModuleShaderFlags::getFunctionFlags(const Function *Func) const { |
| 487 | auto Iter = FunctionFlags.find(Val: Func); |
| 488 | assert((Iter != FunctionFlags.end() && Iter->first == Func) && |
| 489 | "Get Shader Flags : No Shader Flags Mask exists for function" ); |
| 490 | return Iter->second; |
| 491 | } |
| 492 | |
| 493 | //===----------------------------------------------------------------------===// |
| 494 | // ShaderFlagsAnalysis and ShaderFlagsAnalysisPrinterPass |
| 495 | |
| 496 | // Provide an explicit template instantiation for the static ID. |
| 497 | AnalysisKey ShaderFlagsAnalysis::Key; |
| 498 | |
| 499 | ModuleShaderFlags ShaderFlagsAnalysis::run(Module &M, |
| 500 | ModuleAnalysisManager &AM) { |
| 501 | DXILResourceTypeMap &DRTM = AM.getResult<DXILResourceTypeAnalysis>(IR&: M); |
| 502 | DXILResourceMap &DRM = AM.getResult<DXILResourceAnalysis>(IR&: M); |
| 503 | const ModuleMetadataInfo MMDI = AM.getResult<DXILMetadataAnalysis>(IR&: M); |
| 504 | |
| 505 | ModuleShaderFlags MSFI; |
| 506 | MSFI.initialize(M, DRTM, DRM, MMDI); |
| 507 | |
| 508 | return MSFI; |
| 509 | } |
| 510 | |
| 511 | PreservedAnalyses ShaderFlagsAnalysisPrinter::run(Module &M, |
| 512 | ModuleAnalysisManager &AM) { |
| 513 | const ModuleShaderFlags &FlagsInfo = AM.getResult<ShaderFlagsAnalysis>(IR&: M); |
| 514 | // Print description of combined shader flags for all module functions |
| 515 | OS << "; Combined Shader Flags for Module\n" ; |
| 516 | FlagsInfo.getCombinedFlags().print(OS); |
| 517 | // Print shader flags mask for each of the module functions |
| 518 | OS << "; Shader Flags for Module Functions\n" ; |
| 519 | for (const auto &F : M.getFunctionList()) { |
| 520 | if (F.isDeclaration()) |
| 521 | continue; |
| 522 | const ComputedShaderFlags &SFMask = FlagsInfo.getFunctionFlags(Func: &F); |
| 523 | OS << formatv(Fmt: "; Function {0} : {1:x8}\n;\n" , Vals: F.getName(), |
| 524 | Vals: (uint64_t)(SFMask)); |
| 525 | } |
| 526 | |
| 527 | return PreservedAnalyses::all(); |
| 528 | } |
| 529 | |
| 530 | //===----------------------------------------------------------------------===// |
| 531 | // ShaderFlagsAnalysis and ShaderFlagsAnalysisPrinterPass |
| 532 | |
| 533 | bool ShaderFlagsAnalysisWrapper::runOnModule(Module &M) { |
| 534 | DXILResourceTypeMap &DRTM = |
| 535 | getAnalysis<DXILResourceTypeWrapperPass>().getResourceTypeMap(); |
| 536 | DXILResourceMap &DRM = |
| 537 | getAnalysis<DXILResourceWrapperPass>().getResourceMap(); |
| 538 | const ModuleMetadataInfo MMDI = |
| 539 | getAnalysis<DXILMetadataAnalysisWrapperPass>().getModuleMetadata(); |
| 540 | |
| 541 | MSFI.initialize(M, DRTM, DRM, MMDI); |
| 542 | return false; |
| 543 | } |
| 544 | |
| 545 | void ShaderFlagsAnalysisWrapper::getAnalysisUsage(AnalysisUsage &AU) const { |
| 546 | AU.setPreservesAll(); |
| 547 | AU.addRequiredTransitive<DXILResourceTypeWrapperPass>(); |
| 548 | AU.addRequiredTransitive<DXILResourceWrapperPass>(); |
| 549 | AU.addRequired<DXILMetadataAnalysisWrapperPass>(); |
| 550 | } |
| 551 | |
| 552 | char ShaderFlagsAnalysisWrapper::ID = 0; |
| 553 | |
| 554 | INITIALIZE_PASS_BEGIN(ShaderFlagsAnalysisWrapper, "dx-shader-flag-analysis" , |
| 555 | "DXIL Shader Flag Analysis" , true, true) |
| 556 | INITIALIZE_PASS_DEPENDENCY(DXILResourceTypeWrapperPass) |
| 557 | INITIALIZE_PASS_DEPENDENCY(DXILMetadataAnalysisWrapperPass) |
| 558 | INITIALIZE_PASS_END(ShaderFlagsAnalysisWrapper, "dx-shader-flag-analysis" , |
| 559 | "DXIL Shader Flag Analysis" , true, true) |
| 560 | |