diff --git a/docs/ReleaseNotes.md b/docs/ReleaseNotes.md index 4da2a4f644..f261b31861 100644 --- a/docs/ReleaseNotes.md +++ b/docs/ReleaseNotes.md @@ -30,19 +30,19 @@ line upon naming the release. Refer to previous for appropriate section names. available. - Removed work graph support from Shader Model 6.10, and DXIL 1.10 [microsoft/hlsl-specs#915](https://github.com/microsoft/hlsl-specs/issues/915). -- Fixed the set of numeric types allowed in LinAlg matrix intrinsics - [#8271](https://github.com/microsoft/DirectXShaderCompiler/issues/8271). -- Corrected the parameter order of `InterlockedAccumulate` - [microsoft/hlsl-specs#869](https://github.com/microsoft/hlsl-specs/issues/869). -- Added validation of LinAlg matrix builtin parameters and result K dimension - [#8491](https://github.com/microsoft/DirectXShaderCompiler/issues/8491). -- Restricted the component types allowed in LinAlg matrices - [#8494](https://github.com/microsoft/DirectXShaderCompiler/issues/8494). -- Added `BFloat16` to the ComponentType enum in DxilConstants and the linalg - header [#8722](https://github.com/microsoft/DirectXShaderCompiler/issues/8722). #### HLSL Language +- Starting with HLSL 202x, the count in `[unroll(N)]` is a partial-unroll hint + and no longer limits the number of loop iterations + [#8789](https://github.com/microsoft/DirectXShaderCompiler/issues/8789). +- Casting a scalar to a struct or array containing a resource is now an error + instead of crashing + [#6661](https://github.com/microsoft/DirectXShaderCompiler/issues/6661). +- Added the `-Whlsl-2026-compat` warning group for identifying issues + with language changes introduced in HLSL 2026. +- The legacy effects syntax support is removed in HLSL 202x + [#8480](https://github.com/microsoft/DirectXShaderCompiler/issues/8480). - The `shared` and `uniform` keywords are removed in HLSL 202x, with compatibility warnings available for earlier language versions [#8482](https://github.com/microsoft/DirectXShaderCompiler/issues/8482). @@ -56,31 +56,11 @@ line upon naming the release. Refer to previous for appropriate section names. [#8484](https://github.com/microsoft/DirectXShaderCompiler/issues/8484). - Starting with HLSL 202x, `cbuffer` and `tbuffer` declarations and their members belong to their enclosing namespace. -- HLSL 202x supports `const`-qualified instance methods and rejects calls - to non-`const` methods on `const` objects, including objects in constant - buffers - [#8964](https://github.com/microsoft/DirectXShaderCompiler/issues/8964). -- Add `static_assert` matching C++11 and C++17 under HLSL 202x - ([#8910](https://github.com/microsoft/DirectXShaderCompiler/issues/8910)). #### Bug Fixes - Fixed an optimizer crash when scalarizing an out-of-bounds vector access [#8940](https://github.com/microsoft/DirectXShaderCompiler/issues/8940). - -### Upcoming Preview Release - -These changes apply to experimental preview shader models only and will not be -part of the next non-preview release. - -#### Experimental Shader Model 6.11 - -- Added experimental Shader Model 6.11 target profiles. - -### Version 1.9.2609 - -#### Bug Fixes - - Fixed derivative operations being moved into divergent control flow, which could produce incorrect results [#8001](https://github.com/microsoft/DirectXShaderCompiler/issues/8001). @@ -104,22 +84,7 @@ part of the next non-preview release. - Fixed internal compiler errors when a member method is called on a ray payload or on one of its fields with payload access qualifiers enabled [#6464](https://github.com/microsoft/DirectXShaderCompiler/issues/6464). -- Fixed undefined behavior in DXC IntelliSense caused by the uninitialized - `ExpandTokPastingArg` preprocessor option, which made token-pasting - behavior indeterminate. -#### HLSL Language - -- Starting with HLSL 202x, the count in `[unroll(N)]` is a partial-unroll hint - and no longer limits the number of loop iterations - [#8789](https://github.com/microsoft/DirectXShaderCompiler/issues/8789). -- Casting a scalar to a struct or array containing a resource is now an error - instead of crashing - [#6661](https://github.com/microsoft/DirectXShaderCompiler/issues/6661). -- Added the `-Whlsl-2026-compat` warning group for identifying issues - with language changes introduced in HLSL 2026. -- The legacy effects syntax support is removed in HLSL 202x - [#8480](https://github.com/microsoft/DirectXShaderCompiler/issues/8480). #### SPIR-V @@ -138,6 +103,16 @@ part of the next non-preview release. `vk::RawBufferStore` intrinsics [#8572](https://github.com/microsoft/DirectXShaderCompiler/issues/8572). +### Upcoming Preview Release + +These changes apply to experimental preview shader models only and will not be +part of the next non-preview release. + +#### Experimental Shader Model 6.11 + +- Added experimental Shader Model 6.11 target profiles. + + ### Version 1.9.2607 #### HLSL Language diff --git a/include/dxc/dxcapi.internal.h b/include/dxc/dxcapi.internal.h index d256dafac0..ed9a840f83 100644 --- a/include/dxc/dxcapi.internal.h +++ b/include/dxc/dxcapi.internal.h @@ -193,8 +193,6 @@ static const UINT INTRIN_FLAG_READ_ONLY = 1U << 0; static const UINT INTRIN_FLAG_READ_NONE = 1U << 1; static const UINT INTRIN_FLAG_IS_WAVE = 1U << 2; static const UINT INTRIN_FLAG_STATIC_MEMBER = 1U << 3; -// Method mutates the object (cannot be called on a const-qualified instance). -static const UINT INTRIN_FLAG_MUTABLE_METHOD = 1U << 4; struct HLSL_INTRINSIC { UINT Op; // Intrinsic Op ID diff --git a/lib/DxilPIXPasses/DxilAnnotateWithVirtualRegister.cpp b/lib/DxilPIXPasses/DxilAnnotateWithVirtualRegister.cpp index 196da98c87..a596c732e9 100644 --- a/lib/DxilPIXPasses/DxilAnnotateWithVirtualRegister.cpp +++ b/lib/DxilPIXPasses/DxilAnnotateWithVirtualRegister.cpp @@ -106,6 +106,10 @@ class DxilAnnotateWithVirtualRegister : public llvm::ModulePass { m_MST.reset(new llvm::ModuleSlotTracker(&M)); auto functions = m_DM->GetExportedFunctions(); for (auto &fn : functions) { + // A module that names no entry point reports a null one here, e.g. a + // library whose only export is a helper function. + if (fn == nullptr) + continue; m_MST->incorporateFunction(*fn); } } @@ -128,6 +132,12 @@ PrintableSubsetOfMangledFunctionName(llvm::StringRef mangled) { } bool DxilAnnotateWithVirtualRegister::runOnModule(llvm::Module &M) { + // Inline first, so each ordinal this pass hands out belongs to a function + // that PIX can attribute to an invocation. + llvm::SmallVector UninlinedFunctions; + PIXPassHelpers::InlineNonEntryFunctions(M.GetOrCreateDxilModule(), + &UninlinedFunctions); + Init(M); if (m_DM == nullptr) { return false; @@ -218,6 +228,14 @@ bool DxilAnnotateWithVirtualRegister::runOnModule(llvm::Module &M) { } if (OSOverride != nullptr) { + // Name each function that survives inlining. Its instruction range is + // advertised above, but no trace record arrives for it, so PIX must not + // offer it as somewhere to step into. + for (llvm::Function *F : UninlinedFunctions) { + *OSOverride << "UninlinedFunction:" + << PrintableSubsetOfMangledFunctionName(F->getName()) << "\n"; + } + // Print a set of strings of the exemplary form "InstructionCount: // " if (m_DM->GetShaderModel()->GetKind() == hlsl::ShaderModel::Kind::Library) diff --git a/lib/DxilPIXPasses/DxilDbgValueToDbgDeclare.cpp b/lib/DxilPIXPasses/DxilDbgValueToDbgDeclare.cpp index 15a7e6c666..17566217f7 100644 --- a/lib/DxilPIXPasses/DxilDbgValueToDbgDeclare.cpp +++ b/lib/DxilPIXPasses/DxilDbgValueToDbgDeclare.cpp @@ -16,6 +16,7 @@ #include #include "dxc/DXIL/DxilConstants.h" +#include "dxc/DXIL/DxilMetadataHelper.h" #include "dxc/DXIL/DxilModule.h" #include "dxc/DXIL/DxilOperations.h" #include "dxc/DXIL/DxilResourceBase.h" @@ -593,6 +594,19 @@ GlobalStorageMap GatherGlobalEmbeddedArrayStorage(llvm::Module &M) { } bool DxilDbgValueToDbgDeclare::runOnModule(llvm::Module &M) { + // Inline before any shadow storage exists. The stores this pass emits carry + // no debug location on purpose, and llvm::InlineFunction stamps the call site + // location onto each inlined instruction that carries none. Inlining first + // therefore keeps a helper local readable, because its stores stay attributed + // to the helper instead of to the line of the call. + // + // This pass also runs over a plain LLVM module that carries debug info and no + // DXIL, which has no call graph to root the inlining on. + if (M.HasDxilModule() || + M.getNamedMetadata(hlsl::DxilMDHelper::kDxilVersionMDName) != nullptr) { + PIXPassHelpers::InlineNonEntryFunctions(M.GetOrCreateDxilModule()); + } + auto GlobalEmbeddedArrayStorage = GatherGlobalEmbeddedArrayStorage(M); bool Changed = false; diff --git a/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp b/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp index 26f562f180..b7d37bf1dd 100644 --- a/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp +++ b/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp @@ -377,6 +377,28 @@ class DxilDebugInstrumentation : public ModulePass { CountBlockPayloadBytes(std::vector const &IsAndTs); }; +static bool IsInstrumentableShaderKind(DXIL::ShaderKind shaderKind) { + switch (shaderKind) { + case DXIL::ShaderKind::Amplification: + case DXIL::ShaderKind::Mesh: + case DXIL::ShaderKind::Vertex: + case DXIL::ShaderKind::Geometry: + case DXIL::ShaderKind::Pixel: + case DXIL::ShaderKind::Compute: + case DXIL::ShaderKind::RayGeneration: + case DXIL::ShaderKind::Hull: + case DXIL::ShaderKind::Domain: + case DXIL::ShaderKind::Intersection: + case DXIL::ShaderKind::AnyHit: + case DXIL::ShaderKind::ClosestHit: + case DXIL::ShaderKind::Miss: + case DXIL::ShaderKind::Node: + return true; + default: + return false; + } +} + void DxilDebugInstrumentation::applyOptions(PassOptions O) { GetPassOptionUnsigned(O, "FirstInstruction", &m_FirstInstruction, 0); GetPassOptionUnsigned(O, "LastInstruction", &m_LastInstruction, @@ -813,9 +835,17 @@ void DxilDebugInstrumentation::addInvocationSelectionProlog( case DXIL::ShaderKind::Vertex: ParameterTestResult = addVertexShaderProlog(BC, SVIndices); break; - case DXIL::ShaderKind::Hull: - ParameterTestResult = addHullhaderProlog(BC); - break; + case DXIL::ShaderKind::Hull: { + // OutputControlPointID only means something in the control point phase, so + // the patch-constant function is selected by primitive alone. + llvm::Function *function = BC.Builder.GetInsertBlock()->getParent(); + if (function == BC.DM.GetPatchConstantFunction()) { + ParameterTestResult = + addComparePrimitiveIdProlog(BC, m_Parameters.HullShader.PrimitiveId); + } else { + ParameterTestResult = addHullhaderProlog(BC); + } + } break; case DXIL::ShaderKind::Domain: ParameterTestResult = addComparePrimitiveIdProlog(BC, m_Parameters.DomainShader.PrimitiveId); @@ -993,11 +1023,17 @@ uint32_t DxilDebugInstrumentation::addDebugEntryValue(BuilderContext &BC, BC.Builder.CreateFPCast(TheValue, Type::getFloatTy(BC.Ctx), "AsFloat"); BytesToBeEmitted += addDebugEntryValue(BC, AsFloat); } else { + // RawBufferStore is only legal from shader model 6.2 onwards. PIX also + // instruments 6.0 and 6.1 shaders, so fall back to BufferStore (legal from + // 6.0) on those. The two differ only in the trailing alignment operand. + const bool SupportsRawBufferStore = BC.DM.GetShaderModel()->IsSM62Plus(); + const OP::OpCode StoreOpCode = SupportsRawBufferStore + ? OP::OpCode::RawBufferStore + : OP::OpCode::BufferStore; Function *StoreValue = - BC.HlslOP->GetOpFunc(OP::OpCode::RawBufferStore, + BC.HlslOP->GetOpFunc(StoreOpCode, TheValue->getType()); // Type::getInt32Ty(BC.Ctx)); - Constant *StoreValueOpcode = - BC.HlslOP->GetU32Const((unsigned)DXIL::OpCode::RawBufferStore); + Constant *StoreValueOpcode = BC.HlslOP->GetU32Const((unsigned)StoreOpCode); UndefValue *Undef32Arg = UndefValue::get(Type::getInt32Ty(BC.Ctx)); UndefValue *UndefArg = nullptr; if (TheValueTypeID == Type::TypeID::IntegerTyID) { @@ -1014,16 +1050,21 @@ uint32_t DxilDebugInstrumentation::addDebugEntryValue(BuilderContext &BC, auto &values = m_FunctionToValues[BC.Builder.GetInsertBlock()->getParent()]; Constant *RawBufferStoreAlignment = BC.HlslOP->GetU32Const(4); - (void)BC.Builder.CreateCall( - StoreValue, {StoreValueOpcode, // i32 opcode - values.UAVHandle, // %dx.types.Handle, ; resource handle - values.CurrentIndex, // i32 c0: index in bytes into UAV - Undef32Arg, // i32 c1: unused - TheValue, - UndefArg, // unused values - UndefArg, // unused values - UndefArg, // unused values - WriteMask_X, RawBufferStoreAlignment}); + SmallVector StoreArgs{ + StoreValueOpcode, // i32 opcode + values.UAVHandle, // %dx.types.Handle, ; resource handle + values.CurrentIndex, // i32 c0: index in bytes into UAV + Undef32Arg, // i32 c1: unused + TheValue, + UndefArg, // unused values + UndefArg, // unused values + UndefArg, // unused values + WriteMask_X}; + if (SupportsRawBufferStore) { + StoreArgs.push_back(RawBufferStoreAlignment); + } + + (void)BC.Builder.CreateCall(StoreValue, StoreArgs); assert(m_RemainingReservedSpaceInBytes >= 4); // check for underflow m_RemainingReservedSpaceInBytes -= 4; @@ -1313,19 +1354,54 @@ bool DxilDebugInstrumentation::runOnModule(Module &M) { auto ShaderModel = DM.GetShaderModel(); auto shaderKind = ShaderModel->GetKind(); auto HLSLBindId = 0; - auto *uav = PIXPassHelpers::CreateGlobalUAVResource(DM, HLSLBindId, "PIXUAV"); - bool modified = false; + + std::vector functionsToInstrument; if (shaderKind == DXIL::ShaderKind::Library) { - auto instrumentableFunctions = - PIXPassHelpers::GetAllInstrumentableFunctions(DM); - for (auto *F : instrumentableFunctions) { - if (RunOnFunction(M, DM, uav, F)) { - modified = true; - } - } + functionsToInstrument = PIXPassHelpers::GetAllInstrumentableFunctions(DM); } else { + // Only the functions that the runtime itself invokes are instrumented. A + // helper that the entry point calls is not one of them, and cannot become + // one: PIX names an invocation by a record stream in the debug UAV and maps + // that stream to a single function, so instrumenting a helper produces a + // second invocation for one thread whose records PIX then discards. The + // annotation pass inlines such helpers away before anything is numbered. + // See PIXPassHelpers::InlineNonEntryFunctions. llvm::Function *entryFunction = PIXPassHelpers::GetEntryFunction(DM); - modified = RunOnFunction(M, DM, uav, entryFunction); + functionsToInstrument.push_back(entryFunction); + + // The runtime invokes a hull shader patch-constant function rather than the + // entry point does, so it survives inlining and is numbered and advertised + // to PIX as a steppable range of its own. Instrument it too, or a user who + // steps into it sees instructions with no values behind them. + llvm::Function *patchConstantFunction = DM.GetPatchConstantFunction(); + if (patchConstantFunction != nullptr && + patchConstantFunction != entryFunction) { + functionsToInstrument.push_back(patchConstantFunction); + } + } + + functionsToInstrument.erase( + std::remove_if(functionsToInstrument.begin(), functionsToInstrument.end(), + [&DM](llvm::Function *function) { + return function == nullptr || + !IsInstrumentableShaderKind( + PIXPassHelpers::GetFunctionShaderKind( + DM, function)); + }), + functionsToInstrument.end()); + + // Creating the UAV modifies the module, so nothing may be created before the + // pass knows it has something to instrument. + if (functionsToInstrument.empty()) { + return false; + } + + auto *uav = PIXPassHelpers::CreateGlobalUAVResource(DM, HLSLBindId, "PIXUAV"); + bool modified = false; + for (auto *function : functionsToInstrument) { + if (RunOnFunction(M, DM, uav, function)) { + modified = true; + } } return modified; } @@ -1496,23 +1572,7 @@ bool DxilDebugInstrumentation::RunOnFunction(Module &M, DxilModule &DM, DXIL::ShaderKind shaderKind = PIXPassHelpers::GetFunctionShaderKind(DM, function); - switch (shaderKind) { - case DXIL::ShaderKind::Amplification: - case DXIL::ShaderKind::Mesh: - case DXIL::ShaderKind::Vertex: - case DXIL::ShaderKind::Geometry: - case DXIL::ShaderKind::Pixel: - case DXIL::ShaderKind::Compute: - case DXIL::ShaderKind::RayGeneration: - case DXIL::ShaderKind::Hull: - case DXIL::ShaderKind::Domain: - case DXIL::ShaderKind::Intersection: - case DXIL::ShaderKind::AnyHit: - case DXIL::ShaderKind::ClosestHit: - case DXIL::ShaderKind::Miss: - case DXIL::ShaderKind::Node: - break; - default: + if (!IsInstrumentableShaderKind(shaderKind)) { return false; } llvm::SmallPtrSet RayQueryHandles; diff --git a/lib/DxilPIXPasses/DxilPIXDXRInvocationsLog.cpp b/lib/DxilPIXPasses/DxilPIXDXRInvocationsLog.cpp index f456029085..5e6b685554 100644 --- a/lib/DxilPIXPasses/DxilPIXDXRInvocationsLog.cpp +++ b/lib/DxilPIXPasses/DxilPIXDXRInvocationsLog.cpp @@ -67,6 +67,11 @@ bool DxilPIXDXRInvocationsLog::runOnModule(Module &M) { LLVMContext &Ctx = M.getContext(); OP *HlslOP = DM.GetOP(); + // A zero-entry log has no space for records. + if (m_MaxNumEntriesInLog == 0) { + return false; + } + bool Modified = false; for (auto entryFunction : DM.GetExportedFunctions()) { @@ -90,14 +95,11 @@ bool DxilPIXDXRInvocationsLog::runOnModule(Module &M) { dxilutil::FirstNonAllocaInsertionPt(entryFunction); IRBuilder<> Builder(InsertionPoint); - // Add the counter UAV and, when there is space, the record UAV. + // Add the UAVs that we're going to write to CallInst *HandleForCountUAV = PIXPassHelpers::CreateUAVOnceForModule( DM, Builder, /* registerID */ 0, "PIX_CountUAV_Handle"); - CallInst *HandleForUAV = nullptr; - if (m_MaxNumEntriesInLog != 0) { - HandleForUAV = PIXPassHelpers::CreateUAVOnceForModule( - DM, Builder, /* registerID */ 1, "PIX_UAV_Handle"); - } + CallInst *HandleForUAV = PIXPassHelpers::CreateUAVOnceForModule( + DM, Builder, /* registerID */ 1, "PIX_UAV_Handle"); DM.ReEmitDxilResources(); @@ -169,6 +171,15 @@ bool DxilPIXDXRInvocationsLog::runOnModule(Module &M) { Constant *AtomicAdd = HlslOP->GetU32Const((unsigned)DXIL::AtomicBinOpCode::Add); + Function *StoreFuncFloat = + HlslOP->GetOpFunc(OP::OpCode::BufferStore, Type::getFloatTy(Ctx)); + Function *StoreFuncInt = + HlslOP->GetOpFunc(OP::OpCode::BufferStore, Type::getInt32Ty(Ctx)); + Constant *StoreOpcode = + HlslOP->GetU32Const((unsigned)OP::OpCode::BufferStore); + + Constant *WriteMask_XYZW = HlslOP->GetI8Const(15); + Constant *WriteMask_X = HlslOP->GetI8Const(1); Constant *ShaderKindAsConstant = HlslOP->GetU32Const((uint32_t)ShaderKind); Constant *MaxEntryCountAsConstant = HlslOP->GetU32Const((uint32_t)m_MaxNumEntriesInLog); @@ -191,23 +202,9 @@ bool DxilPIXDXRInvocationsLog::runOnModule(Module &M) { }, "EntryIndexResult"); - if (m_MaxNumEntriesInLog == 0) { - continue; - } - - Function *StoreFuncFloat = - HlslOP->GetOpFunc(OP::OpCode::BufferStore, Type::getFloatTy(Ctx)); - Function *StoreFuncInt = - HlslOP->GetOpFunc(OP::OpCode::BufferStore, Type::getInt32Ty(Ctx)); - Constant *StoreOpcode = - HlslOP->GetU32Const((unsigned)OP::OpCode::BufferStore); - - Constant *WriteMask_XYZW = HlslOP->GetI8Const(15); - Constant *WriteMask_X = HlslOP->GetI8Const(1); - // The counter keeps counting past the log capacity. Skip the stores once // the claimed slot is out of range, so the recorded entries stay intact. - Value *EntryIndexIsInRange = Builder.CreateICmpULT( + auto *EntryIndexIsInRange = Builder.CreateICmpULT( EntryIndex, MaxEntryCountAsConstant, "EntryIndexIsInRange"); TerminatorInst *StoreEntryBlockTerminator = SplitBlockAndInsertIfThen(EntryIndexIsInRange, InsertionPoint, @@ -218,13 +215,13 @@ bool DxilPIXDXRInvocationsLog::runOnModule(Module &M) { 4 + (3 * 4) + (3 * 4) + (3 * 4) + 4 + 4 + 4; // See number of bytes we store per shader invocation below - Value *EntryOffset = Builder.CreateMul( + auto EntryOffset = Builder.CreateMul( EntryIndex, HlslOP->GetU32Const(numBytesPerEntry), "EntryOffset"); - Value *EntryOffsetPlus16 = Builder.CreateAdd( + auto EntryOffsetPlus16 = Builder.CreateAdd( EntryOffset, HlslOP->GetU32Const(16), "EntryOffsetPlus16"); - Value *EntryOffsetPlus32 = Builder.CreateAdd( + auto EntryOffsetPlus32 = Builder.CreateAdd( EntryOffset, HlslOP->GetU32Const(32), "EntryOffsetPlus32"); - Value *EntryOffsetPlus48 = Builder.CreateAdd( + auto EntryOffsetPlus48 = Builder.CreateAdd( EntryOffset, HlslOP->GetU32Const(48), "EntryOffsetPlus48"); // Then we start storing the invocation's info into the main UAV buffer diff --git a/lib/DxilPIXPasses/PixPassHelpers.cpp b/lib/DxilPIXPasses/PixPassHelpers.cpp index 9aa14c7846..ebe3701a17 100644 --- a/lib/DxilPIXPasses/PixPassHelpers.cpp +++ b/lib/DxilPIXPasses/PixPassHelpers.cpp @@ -22,6 +22,7 @@ #include "llvm/IR/Module.h" #include "llvm/IR/PassManager.h" #include "llvm/Pass.h" +#include "llvm/Transforms/Utils/Cloning.h" #include "PixPassHelpers.h" @@ -449,6 +450,92 @@ GetAllInstrumentableFunctions(hlsl::DxilModule &DM) { return ret; } +bool InlineNonEntryFunctions( + hlsl::DxilModule &DM, + llvm::SmallVectorImpl *UninlinedFunctions) { + if (UninlinedFunctions != nullptr) { + UninlinedFunctions->clear(); + } + + if (DM.GetShaderModel()->IsLib()) { + return false; + } + + // The runtime invokes a hull shader patch-constant function directly, so it + // is a second root of the call graph and stays alongside the entry point. + llvm::Function *const entryFunction = DM.GetEntryFunction(); + llvm::Function *const patchConstantFunction = DM.GetPatchConstantFunction(); + + // A module that names no entry point has no root, and the entry point has no + // caller in the IR. Leave such a module alone rather than erase every + // function in it. + if (entryFunction == nullptr) { + return false; + } + + bool modified = false; + + // HLSL has no recursion, so the call graph is acyclic and inlining leaf-ward + // terminates. A fixed-point loop also reaches a helper that loses its last + // caller only once another helper is inlined away. + bool inlinedACallThisRound = true; + while (inlinedACallThisRound) { + inlinedACallThisRound = false; + + for (llvm::Function *function : GetAllInstrumentableFunctions(DM)) { + if (function == entryFunction || function == patchConstantFunction) { + continue; + } + + // llvm::InlineFunction is the mechanical inliner and ignores inlining + // attributes. Clear the attribute so the module carries no claim that + // contradicts its own shape. + function->removeFnAttr(llvm::Attribute::NoInline); + + // Collect the call sites first, because inlining rewrites the use list. + llvm::SmallVector callSites; + for (llvm::User *user : function->users()) { + if (auto *call = llvm::dyn_cast(user)) { + if (call->getCalledFunction() == function) { + callSites.push_back(call); + } + } + } + + for (llvm::CallInst *callSite : callSites) { + llvm::InlineFunctionInfo inlineFunctionInfo; + if (llvm::InlineFunction(callSite, inlineFunctionInfo)) { + inlinedACallThisRound = true; + modified = true; + } + } + + // A body with no caller still gets numbered and advertised to PIX as + // somewhere to step into. Erase it. DxilModule keeps an entry-property + // map and a type-annotation map keyed on llvm::Function *, so tell it + // first or both keep entries keyed on freed storage. + if (function->use_empty()) { + DM.RemoveFunction(function); + function->eraseFromParent(); + modified = true; + } + } + } + + if (UninlinedFunctions != nullptr) { + // A function reached other than by a direct call, or one that + // llvm::InlineFunction declines, is still here. PIX gets an instruction + // range for it that no trace record arrives for, so report it. + for (llvm::Function *function : GetAllInstrumentableFunctions(DM)) { + if (function != entryFunction && function != patchConstantFunction) { + UninlinedFunctions->push_back(function); + } + } + } + + return modified; +} + hlsl::DXIL::ShaderKind GetFunctionShaderKind(hlsl::DxilModule &DM, llvm::Function *fn) { hlsl::DXIL::ShaderKind shaderKind = hlsl::DXIL::ShaderKind::Invalid; diff --git a/lib/DxilPIXPasses/PixPassHelpers.h b/lib/DxilPIXPasses/PixPassHelpers.h index 32dd0c1bf5..7302284f2e 100644 --- a/lib/DxilPIXPasses/PixPassHelpers.h +++ b/lib/DxilPIXPasses/PixPassHelpers.h @@ -56,6 +56,36 @@ bool eraseIfUnused(hlsl::DxilModule &DM, llvm::Function *OpFunction); void ClearViewIdState(hlsl::DxilModule &DM); std::vector GetAllInstrumentableFunctions(hlsl::DxilModule &DM); +// Inlines each function that the runtime does not invoke into its callers, and +// erases the inlined-away body. +// +// PIX identifies one shader invocation by one record stream in the debug UAV, +// and maps that stream to exactly one function. A helper instrumented as a +// function of its own therefore reads as a second invocation of a thread that +// runs once, and PIX discards its records. An inlined helper stays visible in +// the inlinedAt chain of the debug locations, which is where PIX looks for it. +// +// Call this before any pass numbers instructions or synthesizes shadow storage. +// PIX steps through the ordinals of the module this leaves behind, and +// llvm::InlineFunction stamps the call site debug location onto each inlined +// instruction that carries none. This function is idempotent, so every pass +// that can come first in a PIX pipeline calls it. +// +// A library module keeps every function, because each exported function is an +// invocation of its own. +// +// UninlinedFunctions, when supplied, receives each non-entry function that is +// still in the module afterwards. Such a function keeps an instruction range +// that no trace record arrives for, so the pass that advertises those ranges +// supplies this parameter and reports what it receives. A pass that advertises +// no range supplies nothing and stays silent, which also keeps one pipeline +// from naming the same function twice. +// +// The survivor set is recomputed on every call, so a caller still receives it +// when an earlier caller already inlined the module. +bool InlineNonEntryFunctions( + hlsl::DxilModule &DM, + llvm::SmallVectorImpl *UninlinedFunctions = nullptr); hlsl::DXIL::ShaderKind GetFunctionShaderKind(hlsl::DxilModule &DM, llvm::Function *fn); #ifdef PIX_DEBUG_DUMP_HELPER diff --git a/tools/clang/include/clang/Basic/DiagnosticParseKinds.td b/tools/clang/include/clang/Basic/DiagnosticParseKinds.td index ef6c15e62a..0e5e5aeca8 100644 --- a/tools/clang/include/clang/Basic/DiagnosticParseKinds.td +++ b/tools/clang/include/clang/Basic/DiagnosticParseKinds.td @@ -1022,9 +1022,6 @@ def err_hlsl_expected_hlsl_attribute : Error < "Unexpected '(' in semantic annotation. Did you mean 'packoffset()' or 'register()'?">; def err_hlsl_enum : Error< "enum is unsupported in HLSL before 2017">; -def err_hlsl_const_member_function_202x - : Error<"const-qualified member functions are unsupported in HLSL before " - "202x">; def warn_hlsl_new_feature : Warning < "%0 is a HLSL %1 feature, and is available in older versions as a non-portable extension.">; diff --git a/tools/clang/include/clang/Basic/DiagnosticSemaKinds.td b/tools/clang/include/clang/Basic/DiagnosticSemaKinds.td index b31ebc6077..650bce5874 100644 --- a/tools/clang/include/clang/Basic/DiagnosticSemaKinds.td +++ b/tools/clang/include/clang/Basic/DiagnosticSemaKinds.td @@ -7926,9 +7926,8 @@ def err_hlsl_unsupported_object_context "entry function parameters|entry function return type|" "patch constant function parameters|patch constant function return type|" "payload parameters|attributes|builtin template parameters|structured buffers|global variables|groupshared variables}1">; -def err_hlsl_unsupported_declaration_in_buffer - : Error<"unsupported declaration %0 in %select{tbuffer|cbuffer}1 " - "declaration">; +def err_hlsl_unsupported_declaration_in_buffer : Error< + "unsupported declaration %0 in %select{tbuffer|cbuffer}1 declaration">; def err_hlsl_logical_binop_scalar : Error< "operands for short-circuiting logical binary operator must be scalar, for non-scalar types use '%select{and|or}0'">; def err_hlsl_ternary_scalar : Error< @@ -8082,9 +8081,6 @@ def err_hlsl_linalg_matrix_attribute_on_invalid_type : Error<"matrix attributes can only be applied to %0">; def err_hlsl_linalg_attributed_matrix_required : Error<"argument must be linear algebra matrix type">; -def err_hlsl_linalg_matrix_global_not_static - : Error<"global variable %0 containing a linear algebra matrix must be " - "declared 'static'">; def err_hlsl_linalg_unsupported_stage : Error< "builtin unavailable in shader stage '%0' (requires 'compute', 'mesh' or 'amplification')">; diff --git a/tools/clang/include/clang/Basic/TokenKinds.def b/tools/clang/include/clang/Basic/TokenKinds.def index 565f74310d..d396427e30 100644 --- a/tools/clang/include/clang/Basic/TokenKinds.def +++ b/tools/clang/include/clang/Basic/TokenKinds.def @@ -349,7 +349,7 @@ CXX11_KEYWORD(constexpr , 0) CXX11_KEYWORD(decltype , 0) CXX11_KEYWORD(noexcept , 0) CXX11_KEYWORD(nullptr , 0) -CXX11_KEYWORD(static_assert , KEYHLSL2026) // HLSL Change - 2026 adds static_assert +CXX11_KEYWORD(static_assert , 0) CXX11_KEYWORD(thread_local , 0) // C++ concepts TS keywords diff --git a/tools/clang/include/clang/Lex/PreprocessorOptions.h b/tools/clang/include/clang/Lex/PreprocessorOptions.h index d9499ea834..6173467c32 100644 --- a/tools/clang/include/clang/Lex/PreprocessorOptions.h +++ b/tools/clang/include/clang/Lex/PreprocessorOptions.h @@ -149,9 +149,7 @@ class PreprocessorOptions : public RefCountedBase { public: PreprocessorOptions() : UsePredefines(true), DetailedRecord(false), - // HLSL Change Begin - ignore line directives. - IgnoreLineDirectives(false), ExpandTokPastingArg(false), - // HLSL Change End + IgnoreLineDirectives(false), // HLSL Change - ignore line directives. DisablePCHValidation(false), AllowPCHWithCompilerErrors(false), DumpDeserializedPCHDecls(false), PrecompiledPreambleBytes(0, true), RemappedFilesKeepOriginalName(true), RetainRemappedFileBuffers(false), diff --git a/tools/clang/lib/AST/Decl.cpp b/tools/clang/lib/AST/Decl.cpp index 134a86de82..79ba3686e5 100644 --- a/tools/clang/lib/AST/Decl.cpp +++ b/tools/clang/lib/AST/Decl.cpp @@ -1958,8 +1958,11 @@ VarDecl::isThisDeclarationADefinition(ASTContext &C) const { getTemplateSpecializationKind() != TSK_ExplicitSpecialization) return DeclarationOnly; - if (hasExternalStorage()) - return DeclarationOnly; + if (!getASTContext() + .getLangOpts() + .HLSL) // HLSL Change - take extern as define to match fxc. + if (hasExternalStorage()) + return DeclarationOnly; // [dcl.link] p7: // A declaration directly contained in a linkage-specification is treated diff --git a/tools/clang/lib/AST/HlslTypes.cpp b/tools/clang/lib/AST/HlslTypes.cpp index 4e82490918..665e81ce30 100644 --- a/tools/clang/lib/AST/HlslTypes.cpp +++ b/tools/clang/lib/AST/HlslTypes.cpp @@ -97,11 +97,9 @@ bool IsHLSLNumericOrAggregateOfNumericType(clang::QualType type) { } // Chars can only appear as part of strings, which we don't consider numeric. - // LinAlg matrix handles are opaque objects, not numeric data. const BuiltinType *BuiltinTy = dyn_cast(Ty); return BuiltinTy != nullptr && - BuiltinTy->getKind() != BuiltinType::Kind::Char_S && - BuiltinTy->getKind() != BuiltinType::Kind::LinAlgMatrix; + BuiltinTy->getKind() != BuiltinType::Kind::Char_S; } // In some cases we need record types that are annotatable and trivially diff --git a/tools/clang/lib/Basic/IdentifierTable.cpp b/tools/clang/lib/Basic/IdentifierTable.cpp index 15fb287ba6..1b27ab39e2 100644 --- a/tools/clang/lib/Basic/IdentifierTable.cpp +++ b/tools/clang/lib/Basic/IdentifierTable.cpp @@ -113,7 +113,6 @@ enum { KEYZVECTOR = 0x40000, KEYHLSL = 0x80000, // MS Change: Flag for hlsl keywords KEYHLSL2026_REMOVED = 0x100000, - KEYHLSL2026 = 0x200000, KEYALL = (0x7ffff & ~KEYNOMS18 & ~KEYNOOPENCL) // KEYNOMS18 and KEYNOOPENCL are used to exclude. }; @@ -143,17 +142,13 @@ static KeywordStatus getKeywordStatus(const LangOptions &LangOpts, if (LangOpts.WChar && (Flags & WCHARSUPPORT)) return KS_Enabled; if (LangOpts.AltiVec && (Flags & KEYALTIVEC)) return KS_Enabled; if (LangOpts.OpenCL && (Flags & KEYOPENCL)) return KS_Enabled; - if (!LangOpts.CPlusPlus && (Flags & KEYNOCXX)) return KS_Enabled; - // HLSL Change Begin - Support for HLSL Keywords. + if (!LangOpts.CPlusPlus && (Flags & KEYNOCXX)) + return KS_Enabled; if (LangOpts.HLSL && LangOpts.HLSLVersion >= hlsl::LangStd::v202x && (Flags & KEYHLSL2026_REMOVED)) return KS_Disabled; - if (LangOpts.HLSL && LangOpts.HLSLVersion >= hlsl::LangStd::v202x && - (Flags & KEYHLSL2026)) - return KS_Enabled; if (LangOpts.HLSL && (Flags & KEYHLSL)) - return KS_Enabled; - // HLSL Change - End + return KS_Enabled; // HLSL Change: Support for HLSL Keywords if (LangOpts.C11 && (Flags & KEYC11)) return KS_Enabled; // We treat bridge casts as objective-C keywords so we can warn on them // in non-arc mode. diff --git a/tools/clang/lib/Headers/hlsl/dx/linalg.h b/tools/clang/lib/Headers/hlsl/dx/linalg.h index ff5e119fa3..d9269f4104 100644 --- a/tools/clang/lib/Headers/hlsl/dx/linalg.h +++ b/tools/clang/lib/Headers/hlsl/dx/linalg.h @@ -126,43 +126,17 @@ struct MatrixLayout { using MatrixLayoutEnum = MatrixLayout::MatrixLayoutEnum; namespace __detail { -template struct ComponentTypeTraits { +template struct ComponentTypeTraits { using Type = uint; static const bool IsNativeScalar = false; static const uint ElementsPerScalar = 4; }; -template struct TypeTraits { +template struct TypeTraits { static const ComponentEnum CompType = (ComponentEnum)dxil::ComponentType::Invalid; }; -template struct IsComponentTypeAvailable { - static const bool value = true; -}; - -#if !__HLSL_ENABLE_16_BIT -template <> struct IsComponentTypeAvailable { - static const bool value = false; -}; -template <> struct IsComponentTypeAvailable { - static const bool value = false; -}; -template <> struct IsComponentTypeAvailable { - static const bool value = false; -}; -#endif - -template struct IsCompatibleVectorElement { - static const bool IsPackedCarrier = - hlsl::is_same::value || - hlsl::is_same::value; - static const bool value = - IsComponentTypeAvailable::value && - (hlsl::is_same::Type>::value || - (!ComponentTypeTraits::IsNativeScalar && IsPackedCarrier)); -}; - template <> struct ComponentTypeTraits { using Type = uint; static const bool IsNativeScalar = false; @@ -192,13 +166,13 @@ __MATRIX_SCALAR_COMPONENT_MAPPING(ComponentType::I64, int64_t) __MATRIX_SCALAR_COMPONENT_MAPPING(ComponentType::U64, uint64_t) __MATRIX_SCALAR_COMPONENT_MAPPING(ComponentType::F64, double) -template struct DstN { +template struct DstN { // Make sure to round up in case SrcN isn't an even multiple of the number of // elements per scalar static const int Value = - (SrcN * ComponentTypeTraits::ElementsPerScalar + - ComponentTypeTraits::ElementsPerScalar - 1) / - ComponentTypeTraits::ElementsPerScalar; + (SrcN * ComponentTypeTraits::ElementsPerScalar + + ComponentTypeTraits::ElementsPerScalar - 1) / + ComponentTypeTraits::ElementsPerScalar; }; template struct DimMN { @@ -211,18 +185,19 @@ template struct DimMN { static const SIZE_TYPE N = MVal; }; -template +template struct ScalarCountFromPackedComponents { static const SIZE_TYPE ElementsPerScalar = - ComponentTypeTraits::ElementsPerScalar; + ComponentTypeTraits::ElementsPerScalar; static const SIZE_TYPE Value = (PackedComponentCount + ElementsPerScalar - 1) / ElementsPerScalar; }; -template struct DefaultAlign { +template +struct DefaultAlign { enum { MinDim = M < N ? M : N, - ScalarCount = ScalarCountFromPackedComponents::Value, + ScalarCount = ScalarCountFromPackedComponents::Value, ByteAlign = ScalarCount * 4, MinByteAlign = ByteAlign < 4 ? 4 : ByteAlign, Value = MinByteAlign < 16 ? MinByteAlign : 16 @@ -231,71 +206,66 @@ template struct DefaultAlign { } // namespace __detail -template struct VectorRef { +template struct VectorRef { ByteAddressBuffer Buf; uint Offset; }; -template struct InterpretedVector { - vector Data; - static const ComponentEnum Interpretation = CT; +template struct InterpretedVector { + vector Data; + static const ComponentEnum Interpretation = DT; static const SIZE_TYPE Size = - __detail::ComponentTypeTraits::ElementsPerScalar * N; + __detail::ComponentTypeTraits
::ElementsPerScalar * N; }; -template -typename hlsl::enable_if< __detail::IsCompatibleVectorElement::value, - InterpretedVector >::type -MakeInterpretedVector(vector Vec) { - InterpretedVector IV = {Vec}; +template +InterpretedVector MakeInterpretedVector(vector Vec) { + InterpretedVector IV = {Vec}; return IV; } -template +template typename hlsl::enable_if< - DestCT != OriginCT && __detail::IsComponentTypeAvailable::value && - __detail::IsCompatibleVectorElement::value, - InterpretedVector::Type, - __detail::DstN::Value, - DestCT> >::type -Convert(vector Vec) { - vector::Type, - __detail::DstN::Value> + DestTy != OriginTy, + InterpretedVector::Type, + __detail::DstN::Value, + DestTy> >::type +Convert(vector Vec) { + vector::Type, + __detail::DstN::Value> Result; - dx::__builtin_LinAlg_Convert(Result, Vec, OriginCT, DestCT); - return MakeInterpretedVector(Result); + dx::__builtin_LinAlg_Convert(Result, Vec, OriginTy, DestTy); + return MakeInterpretedVector(Result); } -template -typename hlsl::enable_if< - DestCT == OriginCT && - __detail::IsCompatibleVectorElement::value, - InterpretedVector >::type -Convert(vector Vec) { - return MakeInterpretedVector(Vec); +template +typename hlsl::enable_if >::type +Convert(vector Vec) { + return MakeInterpretedVector(Vec); } -template +template class Matrix { - using ElementType = typename __detail::ComponentTypeTraits::Type; + using ElementType = typename __detail::ComponentTypeTraits::Type; // If this isn't a native scalar, we have a type that may pack more than 1 // element in each scalar value. (Ex. 8bit => 4elems, 16bit => 2elems) static const uint ElementsPerScalar = - __detail::ComponentTypeTraits::ElementsPerScalar; + __detail::ComponentTypeTraits::ElementsPerScalar; static const bool IsNativeScalar = - __detail::ComponentTypeTraits::IsNativeScalar; + __detail::ComponentTypeTraits::IsNativeScalar; using HandleT = __builtin_LinAlgMatrix - [[__LinAlgMatrix_Attributes(CT, M, N, Use, Scope)]]; + [[__LinAlgMatrix_Attributes(ComponentTy, M, N, Use, Scope)]]; HandleT __handle; - template - [[nodiscard]] Matrix::M, + [[nodiscard]] Matrix::M, __detail::DimMN::N, NewUse, Scope> Cast() { - Matrix::M, + Matrix::M, __detail::DimMN::N, NewUse, Scope> Result; dx::__builtin_LinAlg_CopyConvertMatrix(Result.__handle, __handle, @@ -303,17 +273,17 @@ class Matrix { return Result; } - template + template [[nodiscard]] static - typename hlsl::enable_if::value, Matrix>::type - Splat(Ty Val) { + typename hlsl::enable_if::value, Matrix>::type + Splat(T Val) { Matrix Result; - dx::__builtin_LinAlg_FillMatrix(Result.__handle, hlsl::is_signed::value, + dx::__builtin_LinAlg_FillMatrix(Result.__handle, hlsl::is_signed::value, Val); return Result; } - template ::Value> + template ::Value> [[nodiscard]] static Matrix Load(ByteAddressBuffer Res, uint StartOffset, uint Stride, MatrixLayoutEnum Layout) { Matrix Result; @@ -322,7 +292,7 @@ class Matrix { return Result; } - template ::Value> + template ::Value> [[nodiscard]] static Matrix Load(RWByteAddressBuffer Res, uint StartOffset, uint Stride, MatrixLayoutEnum Layout) { Matrix Result; @@ -331,14 +301,14 @@ class Matrix { return Result; } - template + template [[nodiscard]] static typename hlsl::enable_if< - (hlsl::is_same::type, + (hlsl::is_same::type, ElementType>::value || - hlsl::is_same::type, + hlsl::is_same::type, uint8_t4_packed>::value), Matrix>::type - Load(groupshared Ty Arr[Size], uint StartIdx, uint Stride, + Load(groupshared T Arr[Size], uint StartIdx, uint Stride, MatrixLayoutEnum Layout) { Matrix Result; dx::__builtin_LinAlg_MatrixLoadFromMemory(Result.__handle, Arr, StartIdx, @@ -346,54 +316,58 @@ class Matrix { return Result; } - template - typename hlsl::enable_if::type + template + typename hlsl::enable_if::type Length() { return dx::__builtin_LinAlg_MatrixLength(__handle); } - template - typename hlsl::enable_if::type + template + typename hlsl::enable_if::type GetCoordinate(uint Index) { return dx::__builtin_LinAlg_MatrixGetCoordinate(__handle, Index); } - template - typename hlsl::enable_if::type + template + typename hlsl::enable_if::type Get(uint Index) { ElementType Result; dx::__builtin_LinAlg_MatrixGetElement(Result, __handle, Index); return Result; } - template - typename hlsl::enable_if::type + template + typename hlsl::enable_if::type Set(uint Index, ElementType Value) { dx::__builtin_LinAlg_MatrixSetElement(__handle, __handle, Index, Value); } - template ::Value> + template ::Value> void Store(RWByteAddressBuffer Res, uint StartOffset, uint Stride, MatrixLayoutEnum Layout) { dx::__builtin_LinAlg_MatrixStoreToDescriptor(__handle, Res, StartOffset, Stride, Layout, Align); } - template + template typename hlsl::enable_if< - (hlsl::is_same::type, + (hlsl::is_same::type, ElementType>::value || - hlsl::is_same::type, + hlsl::is_same::type, uint8_t4_packed>::value), void>::type - Store(groupshared Ty Arr[Size], uint StartIdx, uint Stride, + Store(groupshared T Arr[Size], uint StartIdx, uint Stride, MatrixLayoutEnum Layout) { dx::__builtin_LinAlg_MatrixStoreToMemory(__handle, Arr, StartIdx, Stride, Layout); } // Accumulate methods - template ::Value, + template ::Value, MatrixUseEnum UseLocal = Use> typename hlsl::enable_if::type @@ -403,51 +377,51 @@ class Matrix { __handle, Res, StartOffset, Stride, Layout, Align); } - template + template typename hlsl::enable_if< - hlsl::is_same::type, + hlsl::is_same::type, ElementType>::value && - hlsl::is_arithmetic_vector::value && + hlsl::is_arithmetic_vector::value && Use == MatrixUse::Accumulator && UseLocal == Use, void>::type - InterlockedAccumulate(groupshared Ty Arr[Size], uint StartIdx, uint Stride, + InterlockedAccumulate(groupshared T Arr[Size], uint StartIdx, uint Stride, MatrixLayoutEnum Layout) { dx::__builtin_LinAlg_MatrixAccumulateToMemory(__handle, Arr, StartIdx, Stride, Layout); } - template + template typename hlsl::enable_if< - hlsl::is_same::type, + hlsl::is_same::type, uint8_t4_packed>::value && Use == MatrixUse::Accumulator && UseLocal == Use, void>::type - InterlockedAccumulate(groupshared Ty Arr[Size], uint StartIdx, uint Stride, + InterlockedAccumulate(groupshared T Arr[Size], uint StartIdx, uint Stride, MatrixLayoutEnum Layout) { dx::__builtin_LinAlg_MatrixAccumulateToMemory(__handle, Arr, StartIdx, Stride, Layout); } - template + template typename hlsl::enable_if::type - Accumulate(const Matrix MatrixA) { + Accumulate(const Matrix MatrixA) { dx::__builtin_LinAlg_MatrixAccumulate(__handle, __handle, MatrixA.__handle); } - template + template typename hlsl::enable_if::type - Accumulate(const Matrix MatrixB) { + Accumulate(const Matrix MatrixB) { dx::__builtin_LinAlg_MatrixAccumulate(__handle, __handle, MatrixB.__handle); } - template typename hlsl::enable_if::type - MultiplyAccumulate(const Matrix MatrixA, - const Matrix MatrixB) { + MultiplyAccumulate(const Matrix MatrixA, + const Matrix MatrixB) { dx::__builtin_LinAlg_MatrixMatrixMultiplyAccumulate( __handle, MatrixA.__handle, MatrixB.__handle, __handle); } @@ -455,12 +429,13 @@ class Matrix { // Thread-scope Matrices are read-only. Using a template partial // specialization for this simplifies the SFINAE-foo above. -template -class Matrix { - using ElementType = typename __detail::ComponentTypeTraits::Type; +template +class Matrix { + using ElementType = typename __detail::ComponentTypeTraits::Type; - using HandleT = __builtin_LinAlgMatrix - [[__LinAlgMatrix_Attributes(CT, M, N, Use, MatrixScope::Thread)]]; + using HandleT = __builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes( + ComponentTy, M, N, Use, MatrixScope::Thread)]]; HandleT __handle; template -[[nodiscard]] Matrix -Multiply(const Matrix MatrixA, - const Matrix MatrixB) { - Matrix Result; +[[nodiscard]] Matrix +Multiply(const Matrix MatrixA, + const Matrix MatrixB) { + Matrix Result; dx::__builtin_LinAlg_MatrixMatrixMultiply(Result.__handle, MatrixA.__handle, MatrixB.__handle); return Result; } -template -[[nodiscard]] Matrix -Multiply(const Matrix MatrixA, - const Matrix MatrixB) { - Matrix Result; +template +[[nodiscard]] Matrix +Multiply(const Matrix MatrixA, + const Matrix MatrixB) { + Matrix Result; dx::__builtin_LinAlg_MatrixMatrixMultiply(Result.__handle, MatrixA.__handle, MatrixB.__handle); return Result; } -template -[[nodiscard]] Matrix Multiply( - const Matrix MatrixA, - const Matrix MatrixB) { - Matrix Result; + const Matrix MatrixA, + const Matrix MatrixB) { + Matrix Result; dx::__builtin_LinAlg_MatrixMatrixMultiply(Result.__handle, MatrixA.__handle, MatrixB.__handle); return Result; } -template -[[nodiscard]] Matrix +template +[[nodiscard]] Matrix Multiply( - const Matrix MatrixA, - const Matrix MatrixB) { - Matrix Result; + const Matrix MatrixA, + const Matrix + MatrixB) { + Matrix Result; dx::__builtin_LinAlg_MatrixMatrixMultiply(Result.__handle, MatrixA.__handle, MatrixB.__handle); return Result; @@ -538,177 +515,176 @@ Multiply( // Cooperative Vector operates on per-thread vectors multiplying against B // matrices with thread scope. -template -typename hlsl::enable_if::value, - vector >::type -Multiply(Matrix MatrixA, - vector Vec) { - vector Result; +template +typename hlsl::enable_if::value, + vector >::type +Multiply(Matrix MatrixA, + vector Vec) { + vector Result; dx::__builtin_LinAlg_MatrixVectorMultiply( - Result, MatrixA.__handle, hlsl::is_signed::value, Vec, - __detail::TypeTraits::CompType); + Result, MatrixA.__handle, hlsl::is_signed::value, Vec, + __detail::TypeTraits::CompType); return Result; } -template +template typename hlsl::enable_if< - InterpretedVector::Size == K && - __detail::IsCompatibleVectorElement::value, - vector >::type -Multiply(Matrix MatrixA, - InterpretedVector InterpVec) { - vector Result; + InterpretedVector::Size == K, + vector >::type +Multiply(Matrix MatrixA, + InterpretedVector InterpVec) { + vector Result; dx::__builtin_LinAlg_MatrixVectorMultiply( - Result, MatrixA.__handle, hlsl::is_signed::value, + Result, MatrixA.__handle, hlsl::is_signed::value, InterpVec.Data, InterpVec.Interpretation); return Result; } -template -typename hlsl::enable_if::value && - hlsl::is_arithmetic::value, - vector >::type -MultiplyAdd(Matrix MatrixA, - vector Vec, vector Bias) { +template +typename hlsl::enable_if::value && + hlsl::is_arithmetic::value, + vector >::type +MultiplyAdd(Matrix MatrixA, + vector Vec, vector Bias) { - InterpretedVector::CompType> - BiasConvInterp = Convert<__detail::TypeTraits::CompType, - __detail::TypeTraits::CompType>(Bias); + InterpretedVector::CompType> + BiasConvInterp = Convert<__detail::TypeTraits::CompType, + __detail::TypeTraits::CompType>(Bias); - vector Result; + vector Result; dx::__builtin_LinAlg_MatrixVectorMultiplyAdd( - Result, MatrixA.__handle, hlsl::is_signed::value, Vec, - __detail::TypeTraits::CompType, BiasConvInterp.Data); + Result, MatrixA.__handle, hlsl::is_signed::value, Vec, + __detail::TypeTraits::CompType, BiasConvInterp.Data); return Result; } -template +template typename hlsl::enable_if< - VecK == __detail::ScalarCountFromPackedComponents::Value && - __detail::IsCompatibleVectorElement::value && - hlsl::is_arithmetic::value, - vector >::type -MultiplyAdd(Matrix MatrixA, - InterpretedVector InterpVec, - vector Bias) { - - InterpretedVector::CompType> - BiasConvInterp = Convert<__detail::TypeTraits::CompType, - __detail::TypeTraits::CompType>(Bias); - - vector Result; + VecK == __detail::ScalarCountFromPackedComponents::Value && + hlsl::is_arithmetic::value, + vector >::type +MultiplyAdd(Matrix MatrixA, + InterpretedVector InterpVec, + vector Bias) { + + InterpretedVector::CompType> + BiasConvInterp = Convert<__detail::TypeTraits::CompType, + __detail::TypeTraits::CompType>(Bias); + + vector Result; dx::__builtin_LinAlg_MatrixVectorMultiplyAdd( - Result, MatrixA.__handle, hlsl::is_signed::value, + Result, MatrixA.__handle, hlsl::is_signed::value, InterpVec.Data, InterpVec.Interpretation, BiasConvInterp.Data); return Result; } -template -typename hlsl::enable_if::value, - vector >::type -MultiplyAdd(Matrix MatrixA, - vector Vec, VectorRef BiasRef) { +template +typename hlsl::enable_if::value, + vector >::type +MultiplyAdd(Matrix MatrixA, + vector Vec, VectorRef BiasRef) { using BiasVecTy = - vector::Type, - __detail::ScalarCountFromPackedComponents::Value>; + vector::Type, + __detail::ScalarCountFromPackedComponents::Value>; BiasVecTy Bias = BiasRef.Buf.template Load(BiasRef.Offset); - // Convert currently does not support packed type vector sizes that + // FIXME: Convert currently does not support packed type vector sizes that // are not a multiple of the number of elements per scalar, so we // need to do an extra conversion here to get it into the right shape. - // For example, if BiasRef is F8_E4M3FN and M is 7, it gets loaded into - // vector, and if OutputTy is half, Convert will return + // For example if BiasRef is F8_E4M3FN and M is 7, it gets loaded to into + // vector, and if OutputElTy is half, Convert will return // vector instead of vector. // https://github.com/microsoft/DirectXShaderCompiler/issues/8418 - // - // Convert to OutputTy vector with padding + + // Convert to OutputElTy vector with padding using BiasConvInterpPaddedTy = InterpretedVector< - OutputTy, - __detail::DstN< - __detail::TypeTraits::CompType, BiasCT, - __detail::ScalarCountFromPackedComponents< BiasCT, M>::Value>::Value, - __detail::TypeTraits::CompType>; + OutputElTy, + __detail::DstN<__detail::TypeTraits::CompType, BiasElTy, + __detail::ScalarCountFromPackedComponents< + BiasElTy, M>::Value>::Value, + __detail::TypeTraits::CompType>; BiasConvInterpPaddedTy BiasConvInterpPadded = - Convert<__detail::TypeTraits::CompType, BiasCT>(Bias); + Convert<__detail::TypeTraits::CompType, BiasElTy>(Bias); // Truncate the vector to the correct size M - vector BiasConv = (vector)BiasConvInterpPadded.Data; + vector BiasConv = + (vector)BiasConvInterpPadded.Data; - vector Result; + vector Result; dx::__builtin_LinAlg_MatrixVectorMultiplyAdd( - Result, MatrixA.__handle, hlsl::is_signed::value, Vec, - __detail::TypeTraits::CompType, BiasConv); + Result, MatrixA.__handle, hlsl::is_signed::value, Vec, + __detail::TypeTraits::CompType, BiasConv); return Result; } -template +template typename hlsl::enable_if< - VecK == __detail::ScalarCountFromPackedComponents::Value && - __detail::IsCompatibleVectorElement::value, - vector >::type -MultiplyAdd(Matrix MatrixA, - InterpretedVector InterpVec, - VectorRef BiasRef) { + VecK == __detail::ScalarCountFromPackedComponents::Value, + vector >::type +MultiplyAdd(Matrix MatrixA, + InterpretedVector InterpVec, + VectorRef BiasRef) { using BiasVecTy = - vector::Type, - __detail::ScalarCountFromPackedComponents::Value>; + vector::Type, + __detail::ScalarCountFromPackedComponents::Value>; BiasVecTy Bias = BiasRef.Buf.template Load(BiasRef.Offset); - // Convert currently does not support packed type vector sizes that + // FIXME: Convert currently does not support packed type vector sizes that // are not a multiple of the number of elements per scalar, so we // need to do an extra conversion here to get it into the right shape. - // For example, if BiasRef is F8_E4M3FN and M is 7, it gets loaded into - // vector, and if OutputTy is half, Convert will return + // For example if BiasRef is F8_E4M3FN and M is 7, it gets loaded to into + // vector, and if OutputElTy is half, Convert will return // vector instead of vector. // https://github.com/microsoft/DirectXShaderCompiler/issues/8418 - // - // Convert to OutputTy vector with padding + + // Convert to OutputElTy vector with padding using BiasConvInterpPaddedTy = InterpretedVector< - OutputTy, - __detail::DstN< - __detail::TypeTraits::CompType, BiasCT, - __detail::ScalarCountFromPackedComponents< BiasCT, M>::Value>::Value, - __detail::TypeTraits::CompType>; + OutputElTy, + __detail::DstN<__detail::TypeTraits::CompType, BiasElTy, + __detail::ScalarCountFromPackedComponents< + BiasElTy, M>::Value>::Value, + __detail::TypeTraits::CompType>; BiasConvInterpPaddedTy BiasConvInterpPadded = - Convert<__detail::TypeTraits::CompType, BiasCT>(Bias); + Convert<__detail::TypeTraits::CompType, BiasElTy>(Bias); // Truncate the vector to the correct size M - vector BiasConv = (vector)BiasConvInterpPadded.Data; + vector BiasConv = + (vector)BiasConvInterpPadded.Data; - vector Result; + vector Result; dx::__builtin_LinAlg_MatrixVectorMultiplyAdd( - Result, MatrixA.__handle, hlsl::is_signed::value, + Result, MatrixA.__handle, hlsl::is_signed::value, InterpVec.Data, InterpVec.Interpretation, BiasConv); return Result; } // Outer product functions -template +template [[nodiscard]] typename hlsl::enable_if< - hlsl::is_arithmetic::value, - Matrix >::type -OuterProduct(vector VecA, vector VecB) { - Matrix Result; + hlsl::is_arithmetic::value, + Matrix >::type +OuterProduct(vector VecA, vector VecB) { + Matrix Result; dx::__builtin_LinAlg_MatrixOuterProduct( - Result.__handle, hlsl::is_signed::value, VecA, VecB); + Result.__handle, hlsl::is_signed::value, VecA, VecB); return Result; } -template -typename hlsl::enable_if::value, void>::type +template +typename hlsl::enable_if::value, void>::type InterlockedAccumulate(RWByteAddressBuffer Res, uint StartOffset, - vector Vec) { + vector Vec) { dx::__builtin_LinAlg_VectorAccumulateToDescriptor(Res, StartOffset, Align, Vec); } diff --git a/tools/clang/lib/Parse/ParseDecl.cpp b/tools/clang/lib/Parse/ParseDecl.cpp index eea8e58d89..63d1562cf0 100644 --- a/tools/clang/lib/Parse/ParseDecl.cpp +++ b/tools/clang/lib/Parse/ParseDecl.cpp @@ -6551,28 +6551,16 @@ void Parser::ParseFunctionDeclarator(Declarator &D, // with the pure-specifier in the same way. // Parse cv-qualifier-seq[opt]. - // HLSL Change Starts - // HLSL only supports `const` here (HLSL 202x). Parse it directly because - // the HLSL path of ParseTypeQualifierListOpt is shared with other - // declarator contexts (e.g. array bounds) that must still reject it. - if (getLangOpts().HLSL) { - while (Tok.is(tok::kw_const)) { - const char *PrevSpec = nullptr; - unsigned DiagID = 0; - SourceLocation Loc = Tok.getLocation(); - if (DS.SetTypeQual(DeclSpec::TQ_const, Loc, PrevSpec, DiagID, - getLangOpts())) - Diag(Loc, DiagID) << PrevSpec; - DS.SetRangeEnd(ConsumeToken()); - } - if (DS.getConstSpecLoc().isValid() && - getLangOpts().HLSLVersion < hlsl::LangStd::v202x) - Diag(DS.getConstSpecLoc(), diag::err_hlsl_const_member_function_202x); - } - // HLSL Change Ends ParseTypeQualifierListOpt(DS, AR_NoAttributesParsed, /*AtomicAllowed*/ false); if (!DS.getSourceRange().getEnd().isInvalid()) { + // HLSL Change Starts + if (getLangOpts().HLSL) { + Diag(DS.getSourceRange().getEnd(), + diag::err_hlsl_unsupported_construct) + << "qualifiers"; + } + // HLSL Change Ends EndLoc = DS.getSourceRange().getEnd(); ConstQualifierLoc = DS.getConstSpecLoc(); VolatileQualifierLoc = DS.getVolatileSpecLoc(); diff --git a/tools/clang/lib/Parse/ParseDeclCXX.cpp b/tools/clang/lib/Parse/ParseDeclCXX.cpp index 0861cf2de5..af178ae8eb 100644 --- a/tools/clang/lib/Parse/ParseDeclCXX.cpp +++ b/tools/clang/lib/Parse/ParseDeclCXX.cpp @@ -736,18 +736,12 @@ Decl *Parser::ParseStaticAssertDeclaration(SourceLocation &DeclEnd){ ExprResult AssertMessage; if (Tok.is(tok::r_paren)) { - // HLSL Change Starts - In HLSL 202x, allow omitting the message just like - // C++17 does, without emitting an extension warning. - bool AllowNoMessage = getLangOpts().CPlusPlus1z || - (getLangOpts().HLSL && - getLangOpts().HLSLVersion >= hlsl::LangStd::v202x); - if (!AllowNoMessage) - Diag(Tok, diag::ext_static_assert_no_message) - << FixItHint::CreateInsertion(Tok.getLocation(), ", \"\""); - else if (getLangOpts().CPlusPlus1z) - Diag(Tok, diag::warn_cxx14_compat_static_assert_no_message) - << FixItHint(); - // HLSL Change Ends + Diag(Tok, getLangOpts().CPlusPlus1z + ? diag::warn_cxx14_compat_static_assert_no_message + : diag::ext_static_assert_no_message) + << (getLangOpts().CPlusPlus1z + ? FixItHint() + : FixItHint::CreateInsertion(Tok.getLocation(), ", \"\"")); } else { if (ExpectAndConsume(tok::comma)) { SkipUntil(tok::semi); diff --git a/tools/clang/lib/SPIRV/SpirvEmitter.cpp b/tools/clang/lib/SPIRV/SpirvEmitter.cpp index 45ec7e75c2..a8c987288f 100644 --- a/tools/clang/lib/SPIRV/SpirvEmitter.cpp +++ b/tools/clang/lib/SPIRV/SpirvEmitter.cpp @@ -1051,7 +1051,7 @@ void SpirvEmitter::HandleTranslationUnit(ASTContext &context) { void SpirvEmitter::doDecl(const Decl *decl) { if (isa(decl) || isa(decl) || - isa(decl) || isa(decl)) + isa(decl)) return; // Implicit decls are lazily created when needed. diff --git a/tools/clang/lib/Sema/SemaDecl.cpp b/tools/clang/lib/Sema/SemaDecl.cpp index b6ed4fe8ca..fe6369ffd5 100644 --- a/tools/clang/lib/Sema/SemaDecl.cpp +++ b/tools/clang/lib/Sema/SemaDecl.cpp @@ -3443,14 +3443,6 @@ void Sema::MergeVarDecl(VarDecl *New, LookupResult &Previous, ShadowMergeState& if (New->isInvalidDecl()) return; - // HLSL does not permit multiple declarations of a global variable. - if (getLangOpts().HLSL && New->isFileVarDecl() && Old->isFileVarDecl() && - !New->isStaticDataMember() && !Old->isStaticDataMember()) { - Diag(New->getLocation(), diag::err_redefinition) << New->getDeclName(); - Diag(Old->getLocation(), diag::note_previous_definition); - return New->setInvalidDecl(); - } - diag::kind PrevDiag; SourceLocation OldLocation; std::tie(PrevDiag, OldLocation) = diff --git a/tools/clang/lib/Sema/SemaHLSL.cpp b/tools/clang/lib/Sema/SemaHLSL.cpp index 5d480a99ee..8a8eb96f62 100644 --- a/tools/clang/lib/Sema/SemaHLSL.cpp +++ b/tools/clang/lib/Sema/SemaHLSL.cpp @@ -2002,15 +2002,6 @@ static bool IsStaticMember(const HLSL_INTRINSIC *fn) { return fn->Flags & INTRIN_FLAG_STATIC_MEMBER; } -// Returns true if the intrinsic is a non-static method that does not mutate -// instance state. Writing through a resource handle does not mutate the handle. -static bool IsConstMemberIntrinsic(const HLSL_INTRINSIC *fn) { - if (IsStaticMember(fn)) - return false; - // A method is const unless it explicitly mutates the object. - return !(fn->Flags & INTRIN_FLAG_MUTABLE_METHOD); -} - static bool IsVariadicIntrinsicFunction(const HLSL_INTRINSIC *fn) { return fn->pArgs[fn->uNumArgs - 1].uTemplateId == INTRIN_TEMPLATE_VARARGS; } @@ -3458,12 +3449,11 @@ class HLSLExternalSource : public ExternalSemaSource { DeclarationName declarationName = DeclarationName(ii); StorageClass SC = IsStaticMember(intrinsic) ? SC_Static : SC_None; - bool IsConst = IsConstMemberIntrinsic(intrinsic); CXXMethodDecl *functionDecl = CreateObjectFunctionDeclarationWithParams( *m_context, recordDecl, functionResultQT, ArrayRef(argsQTs, numParams), - ArrayRef(argNames, numParams), declarationName, IsConst, SC, + ArrayRef(argNames, numParams), declarationName, true, SC, templateParamNamedDeclsCount > 0); functionDecl->setImplicit(true); @@ -5681,8 +5671,6 @@ class HLSLExternalSource : public ExternalSemaSource { /// numeric elements exclusively. bool IsTypeNumeric(QualType type, UINT *count); - bool ContainsLinAlgMatrixType(QualType type); - /// Checks whether the specified type is a scalar type. bool IsScalarType(const QualType &type) { DXASSERT(!type.isNull(), "caller should validate its type is initialized"); @@ -6396,14 +6384,11 @@ class HLSLExternalSource : public ExternalSemaSource { MultiLevelTemplateArgumentList mlTemplateArgumentList(templateArgumentList); TemplateDeclInstantiator declInstantiator(*this->m_sema, owner, mlTemplateArgumentList); - FunctionProtoType::ExtProtoInfo EPI; - // Preserve the method's const qualification on the resolved specialization. - if (IsConstMemberIntrinsic(intrinsic)) - EPI.TypeQuals = Qualifiers::Const; + FunctionProtoType::ExtProtoInfo EmptyEPI; QualType functionType = m_context->getFunctionType( parameterTypes[0], - ArrayRef(parameterTypes + 1, parameterTypeCount - 1), EPI, - paramMods); + ArrayRef(parameterTypes + 1, parameterTypeCount - 1), + EmptyEPI, paramMods); TypeSourceInfo *TInfo = m_context->CreateTypeSourceInfo(functionType, 0); FunctionProtoTypeLoc Proto = TInfo->getTypeLoc().getAs(); @@ -8710,9 +8695,8 @@ UINT64 HLSLExternalSource::ScoreFunction(OverloadCandidateSet::iterator &Cand) { // in/out considerations have been taken care of by viability. - // The implicit object argument (`this`) affects lookup and viability. - // In HLSL 202x, its const qualification also breaks ties between viable - // overloads: a non-const object prefers a non-const method. + // 'this' considerations don't matter without inheritance, other + // than lookup and viability. UINT64 result = 0; for (unsigned convIdx = 0; convIdx < Cand->NumConversions; ++convIdx) { @@ -8730,23 +8714,6 @@ UINT64 HLSLExternalSource::ScoreFunction(OverloadCandidateSet::iterator &Cand) { } result += score; } - - // HLSL 202x: when both const and non-const overloads of a method are - // viable for a non-const object, prefer the non-const overload. Add a - // small tie-breaking penalty when the implicit object argument requires - // adding `const` to call a const-qualified method. This uses the low score - // bits reserved by SCORE_MIN_SHIFT. - CXXMethodDecl *Method = dyn_cast_or_null(Cand->Function); - if (m_sema->getLangOpts().HLSLVersion >= hlsl::LangStd::v202x && Method && - !Cand->IgnoreObjectArgument && - (Method->getTypeQualifiers() & Qualifiers::Const)) { - const ImplicitConversionSequence &ICS = Cand->Conversions[0]; - if (ICS.isStandard()) { - QualType FromType = ICS.Standard.getFromType(); - if (!FromType.isNull() && !FromType.isConstQualified()) - result += 1; - } - } return result; } @@ -9058,40 +9025,10 @@ bool HLSLExternalSource::IsTypeNumeric(QualType type, UINT *count) { case AR_TOBJ_OBJECT: case AR_TOBJ_DEPENDENT: case AR_TOBJ_STRING: - case AR_TOBJ_LINALG_MATRIX: return false; } } -bool HLSLExternalSource::ContainsLinAlgMatrixType(QualType Type) { - DXASSERT_NOMSG(!Type.isNull()); - - Type = GetStructuralForm(Type); - // Covers both attributed matrices and the unattributed builtin handle. - if (Type->isAttributedLinAlgMatrixType() || Type->isLinAlgMatrixType()) - return true; - - if (const ArrayType *AT = m_context->getAsArrayType(Type)) - return ContainsLinAlgMatrixType(AT->getElementType()); - - if (GetTypeObjectKind(Type) != AR_TOBJ_COMPOUND) - return false; - - const CXXRecordDecl *RD = Type->getAsCXXRecordDecl(); - if (!RD || !RD->hasDefinition()) - return false; - - for (const CXXBaseSpecifier &Base : RD->bases()) - if (ContainsLinAlgMatrixType(Base.getType())) - return true; - - for (const FieldDecl *Field : RD->fields()) - if (ContainsLinAlgMatrixType(Field->getType())) - return true; - - return false; -} - enum MatrixMemberAccessError { MatrixMemberAccessError_None, // No errors found. MatrixMemberAccessError_BadFormat, // Formatting error (non-digit). @@ -12837,17 +12774,6 @@ static bool AllowObjectInContext(QualType Ty, TypeDiagContext DiagContext) { return true; } -// LinAlg matrices (attributed or the raw builtin handle) are opaque, thread -// local values. They are only valid as static global state, locals, and -// non-entry function parameters and return types, never in resources, -// groupshared memory, or shader interfaces. -static bool AllowLinAlgMatrixInContext(TypeDiagContext DiagContext) { - // Non-static globals are rejected separately with a diagnostic that asks for - // an explicit 'static'. - return DiagContext == TypeDiagContext::GlobalVariables || - DiagContext == TypeDiagContext::CBuffersOrTBuffers; -} - // Determine if `Ty` is valid in this `DiagContext` and/or an empty type. If // invalid returns false and Sema `S`, location `Loc`, error index // `DiagContext`, and FieldDecl `FD` are used to emit diagnostics. If @@ -12878,21 +12804,6 @@ DiagnoseElementTypes(Sema &S, SourceLocation Loc, QualType Ty, bool &Empty, static_cast(TypeDiagContext::LongVecDiagMaxSelectIndex))); HLSLExternalSource *Source = HLSLExternalSource::FromSema(&S); - - const Type *CanonTy = Ty.getCanonicalType().getTypePtr(); - if (CanonTy->isAttributedLinAlgMatrixType() || - CanonTy->isLinAlgMatrixType()) { - Empty = false; - if (!CheckObjects || AllowLinAlgMatrixInContext(ObjDiagContext)) - return false; - S.Diag(Loc, diag::err_hlsl_unsupported_object_context) - << Ty << ObjDiagContextIdx; - if (FD) - S.Diag(FD->getLocation(), diag::note_field_declared_here) - << FD->getType() << FD->getSourceRange(); - return true; - } - ArTypeObjectKind ShapeKind = Source->GetTypeObjectKind(Ty); switch (ShapeKind) { case AR_TOBJ_VECTOR: @@ -16192,15 +16103,6 @@ bool Sema::DiagnoseHLSLDecl(Declarator &D, DeclContext *DC, Expr *BitWidth, if (DiagnoseTypeElements(*this, D.getLocStart(), qt, ObjDiagContext, LongVecDiagContext)) result = false; - - // LinAlg matrices are mutable state that cannot live in the implicit - // global constant buffer. Groupshared is rejected above. - if (!isStatic && !isGroupShared && !D.isInvalidType() && - !qt->isDependentType() && hlslSource->ContainsLinAlgMatrixType(qt)) { - Diag(D.getLocStart(), diag::err_hlsl_linalg_matrix_global_not_static) - << D.getIdentifier(); - result = false; - } } // SPIRV change starts diff --git a/tools/clang/lib/Sema/SemaOverload.cpp b/tools/clang/lib/Sema/SemaOverload.cpp index b21da0c7e1..3b3f15ff71 100644 --- a/tools/clang/lib/Sema/SemaOverload.cpp +++ b/tools/clang/lib/Sema/SemaOverload.cpp @@ -4870,26 +4870,13 @@ TryObjectArgumentInitialization(Sema &S, QualType FromType, // First check the qualifiers. QualType FromTypeCanon = S.Context.getCanonicalType(FromType); // HLSL Change Starts - // HLSL Note: Prior to HLSL 202x, for calls that aren't compiler-generated - // C++ overloads, we disregard const qualifiers so that member functions can - // be called on `const` objects from constant buffer types. - // - // HLSL 202x supports `const` instance methods. When the method is const- - // qualified, or when the object is const-qualified, we enforce - // const-correctness so that: - // - a const object cannot call a non-const method - // - a non-const object calling a const method incurs a qualification - // adjustment (which the HLSL overload scorer can use to prefer the - // non-const overload). + // HLSL Note: For calls that aren't compiler-generated C++ overloads, we + // disregard const qualifiers so that member functions can be called on + // `const` objects from constant buffer types. This should change in the + // future if we support const instance methods. FromTypeCanon.removeLocalRestrict(); // HLSL Change - disregard restrict. - bool EnforceHLSLConst = S.getLangOpts().HLSL && - S.getLangOpts().HLSLVersion >= hlsl::LangStd::v202x && - !isa(Method) && - ((Method->getTypeQualifiers() & Qualifiers::Const) || - FromTypeCanon.isConstQualified()); if (!S.getLangOpts().HLSL || - (Method != nullptr && Method->hasAttr()) || - EnforceHLSLConst) { + (Method != nullptr && Method->hasAttr())) { // HLSL Change Ends if (ImplicitParamType.getCVRQualifiers() != FromTypeCanon.getLocalCVRQualifiers() && diff --git a/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/clusterid.hlsl b/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/clusterid.hlsl index 11b16a8c7e..a4bb456918 100644 --- a/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/clusterid.hlsl +++ b/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/clusterid.hlsl @@ -5,19 +5,19 @@ // Test ClusterID intrinsics for SM 6.10 -// AST: `-CXXMethodDecl {{.*}} used GetClusterID 'unsigned int () const' extern +// AST: `-CXXMethodDecl {{.*}} used GetClusterID 'unsigned int ()' extern // AST-NEXT: {{.*}}|-TemplateArgument type 'unsigned int' // AST-NEXT: {{.*}}|-HLSLIntrinsicAttr {{.*}} Implicit "op" "" 396 // AST-NEXT: {{.*}}|-ConstAttr {{.*}} Implicit // AST-NEXT: {{.*}}`-AvailabilityAttr {{.*}} Implicit 6.10 0 0 "" -// AST: `-CXXMethodDecl {{.*}} used CandidateClusterID 'unsigned int () const' extern +// AST: `-CXXMethodDecl {{.*}} used CandidateClusterID 'unsigned int ()' extern // AST-NEXT: {{.*}}|-TemplateArgument type 'unsigned int' // AST-NEXT: {{.*}}|-HLSLIntrinsicAttr {{.*}} Implicit "op" "" 394 // AST-NEXT: {{.*}}|-PureAttr {{.*}} Implicit // AST-NEXT: {{.*}}`-AvailabilityAttr {{.*}} Implicit 6.10 0 0 "" -// AST: `-CXXMethodDecl {{.*}} used CommittedClusterID 'unsigned int () const' extern +// AST: `-CXXMethodDecl {{.*}} used CommittedClusterID 'unsigned int ()' extern // AST-NEXT: {{.*}}|-TemplateArgument type 'unsigned int' // AST-NEXT: {{.*}}|-HLSLIntrinsicAttr {{.*}} Implicit "op" "" 395 // AST-NEXT: {{.*}}|-PureAttr {{.*}} Implicit diff --git a/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/triangle_positions.hlsl b/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/triangle_positions.hlsl index 15a2df6825..145c9839d9 100644 --- a/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/triangle_positions.hlsl +++ b/tools/clang/test/CodeGenDXIL/hlsl/intrinsics/triangle_positions.hlsl @@ -3,19 +3,19 @@ // RUN: %dxc -T lib_6_10 %s -ast-dump-implicit | FileCheck %s --check-prefix AST // RUN: %dxc -T lib_6_10 %s -fcgl | FileCheck %s --check-prefix FCGL -// AST: `-CXXMethodDecl {{.*}} <> used TriangleObjectPositions 'BuiltInTrianglePositions &() const' extern +// AST: `-CXXMethodDecl {{.*}} <> used TriangleObjectPositions 'BuiltInTrianglePositions &()' extern // AST-NEXT: |-TemplateArgument type 'BuiltInTrianglePositions' // AST-NEXT: |-HLSLIntrinsicAttr {{.*}} <> Implicit "op" "" 400 // AST-NEXT: |-ConstAttr {{.*}} <> Implicit // AST-NEXT: `-AvailabilityAttr {{.*}} <> Implicit 6.10 0 0 "" -// AST: `-CXXMethodDecl {{.*}} <> used CandidateTriangleObjectPositions 'BuiltInTrianglePositions &() const' extern +// AST: `-CXXMethodDecl {{.*}} <> used CandidateTriangleObjectPositions 'BuiltInTrianglePositions &()' extern // AST: |-TemplateArgument type 'BuiltInTrianglePositions' // AST: |-HLSLIntrinsicAttr {{.*}} <> Implicit "op" "" 398 // AST: |-PureAttr {{.*}} <> Implicit // AST: `-AvailabilityAttr {{.*}} <> Implicit 6.10 0 0 "" -// AST `-CXXMethodDecl {{.*}} <> used CommittedTriangleObjectPositions 'BuiltInTrianglePositions &() const' extern +// AST `-CXXMethodDecl {{.*}} <> used CommittedTriangleObjectPositions 'BuiltInTrianglePositions &()' extern // AST |-TemplateArgument type 'BuiltInTrianglePositions' // AST |-HLSLIntrinsicAttr {{.*}} <> Implicit "op" "" 399 // AST |-PureAttr {{.*}} <> Implicit diff --git a/tools/clang/test/CodeGenDXIL/hlsl/linalg/linalg-matrix-global.hlsl b/tools/clang/test/CodeGenDXIL/hlsl/linalg/linalg-matrix-global.hlsl deleted file mode 100644 index 31b29a9cbe..0000000000 --- a/tools/clang/test/CodeGenDXIL/hlsl/linalg/linalg-matrix-global.hlsl +++ /dev/null @@ -1,39 +0,0 @@ -// REQUIRES: dxil-1-10 -// RUN: %dxc -T cs_6_10 -E main -fcgl %s | FileCheck %s -// RUN: %dxc -T cs_6_10 -E main %s | FileCheck %s --check-prefix=DXIL - -// Explicitly static LinAlg matrix globals are mutable module state, not -// constant buffer data. - -#include -using namespace dx::linalg; - -using MatrixTy = - Matrix; - -struct MatrixState { - MatrixTy Matrix; -}; - -static MatrixTy GlobalMatrix; -static MatrixTy GlobalMatrixArray[2]; -static MatrixState GlobalMatrixState; -RWByteAddressBuffer Output; - -[numthreads(1, 1, 1)] -void main() { - GlobalMatrix = MatrixTy::Splat(1.0f); - Output.Store(0, GlobalMatrix.Get(0)); -} - -// CHECK-NOT: dx.hl.subscript.cb -// CHECK: @GlobalMatrix = internal global -// CHECK-NOT: dx.hl.subscript.cb -// CHECK: bitcast {{.*}} @GlobalMatrix -// CHECK: call float {{.*}} @GlobalMatrix -// CHECK-NOT: dx.hl.subscript.cb - -// DXIL: call %dx.types.LinAlgMatrixC9M4N4U2S1 @dx.op.linAlgFillMatrix -// DXIL: call float @dx.op.linAlgMatrixGetElement -// DXIL-NOT: @dx.op.cbufferLoad diff --git a/tools/clang/test/CodeGenSPIRV/static.assert.hlsl b/tools/clang/test/CodeGenSPIRV/static.assert.hlsl deleted file mode 100644 index 5b4196bbf2..0000000000 --- a/tools/clang/test/CodeGenSPIRV/static.assert.hlsl +++ /dev/null @@ -1,21 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 202x -spirv %s | FileCheck %s - -static_assert(1 == 1, "translation unit"); -static_assert(sizeof(float) == 4); - -namespace N { -static_assert(2 + 2 == 4, "namespace"); -} - -struct S { - static_assert(sizeof(float) == 4, "record"); - float Value; -}; - -float main() : SV_Target { - static_assert(sizeof(S) == 4, "function"); - static_assert(1 < 2); - return 0; -} - -// CHECK: OpEntryPoint Fragment %main "main" diff --git a/tools/clang/test/HLSL/cpp-errors-hv2015.hlsl b/tools/clang/test/HLSL/cpp-errors-hv2015.hlsl index fa77e86d50..57c512741c 100644 --- a/tools/clang/test/HLSL/cpp-errors-hv2015.hlsl +++ b/tools/clang/test/HLSL/cpp-errors-hv2015.hlsl @@ -59,7 +59,7 @@ struct s_with_friend { friend void some_fn(); // expected-error {{'friend' is a reserved keyword in HLSL}} }; -typedef int (*fn_int_const)(int) const; // expected-error {{const-qualified member functions are unsupported in HLSL before 202x}} expected-error {{pointers are unsupported in HLSL}} +typedef int (*fn_int_const)(int) const; // expected-error {{expected ';' after top level declarator}} expected-error {{pointers are unsupported in HLSL}} expected-warning {{declaration does not declare anything}} typedef int (*fn_int_volatile)(int) volatile; // expected-error {{'volatile' is a reserved keyword in HLSL}} expected-error {{expected ';' after top level declarator}} expected-error {{pointers are unsupported in HLSL}} expected-warning {{declaration does not declare anything}} void fn_throw() throw() { } // expected-error {{exception specification is unsupported in HLSL}} diff --git a/tools/clang/test/HLSL/cpp-errors.hlsl b/tools/clang/test/HLSL/cpp-errors.hlsl index 2f23453407..1ecf4a57e1 100644 --- a/tools/clang/test/HLSL/cpp-errors.hlsl +++ b/tools/clang/test/HLSL/cpp-errors.hlsl @@ -56,7 +56,7 @@ struct s_with_friend { friend void some_fn(); // expected-error {{'friend' is a reserved keyword in HLSL}} }; -typedef int (*fn_int_const)(int) const; // expected-error {{const-qualified member functions are unsupported in HLSL before 202x}} expected-error {{pointers are unsupported in HLSL}} +typedef int (*fn_int_const)(int) const; // expected-error {{expected ';' after top level declarator}} expected-error {{pointers are unsupported in HLSL}} expected-warning {{declaration does not declare anything}} typedef int (*fn_int_volatile)(int) volatile; // expected-error {{'volatile' is a reserved keyword in HLSL}} expected-error {{expected ';' after top level declarator}} expected-error {{pointers are unsupported in HLSL}} expected-warning {{declaration does not declare anything}} void fn_throw() throw() { } // expected-error {{exception specification is unsupported in HLSL}} diff --git a/tools/clang/test/HLSLFileCheck/hlsl/classes/const_method_202x_codegen.hlsl b/tools/clang/test/HLSLFileCheck/hlsl/classes/const_method_202x_codegen.hlsl deleted file mode 100644 index 4394df7a01..0000000000 --- a/tools/clang/test/HLSLFileCheck/hlsl/classes/const_method_202x_codegen.hlsl +++ /dev/null @@ -1,28 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 202x %s | FileCheck %s - -// Verify that const-instance methods generate valid DXIL: a const method -// called on a non-const local lvalue should be inlined as a normal read of -// the object's fields, and a const method called on a cbuffer member should -// lower to cbufferLoadLegacy. - -struct S { - int x; - int y; - int sum() const { return x + y; } -}; - -cbuffer CB { S cs; }; - -int main(int idx : A) : SV_Target { - S ls = {3, 4}; - return ls.sum() + cs.sum(); -} - -// CHECK: define void @main() -// CHECK: call %dx.types.Handle @dx.op.createHandle( -// CHECK: call %dx.types.CBufRet.i32 @dx.op.cbufferLoadLegacy.i32( -// The 3+4 from the local 'ls' is constant-folded to 7 and added to the -// two i32 lanes loaded from the cbuffer. -// CHECK: add i32 {{.*}}, 7 -// CHECK: call void @dx.op.storeOutput.i32( -// CHECK: ret void diff --git a/tools/clang/test/HLSLFileCheck/hlsl/classes/const_resource_local.hlsl b/tools/clang/test/HLSLFileCheck/hlsl/classes/const_resource_local.hlsl deleted file mode 100644 index 24e0c0a76b..0000000000 --- a/tools/clang/test/HLSLFileCheck/hlsl/classes/const_resource_local.hlsl +++ /dev/null @@ -1,41 +0,0 @@ -// RUN: %dxc -T ps_6_6 -E main -HV 202x %s | FileCheck %s - -// Verify that const local resource objects can still be used through their -// instance methods (which are now properly marked const). A const handle only -// prevents reassigning the handle, so writing through a const RW resource is -// still allowed. - -Texture2D tex : register(t0); -SamplerState samp : register(s0); -RWBuffer buf : register(u0); -ByteAddressBuffer bab : register(t1); -StructuredBuffer sb : register(t2); - -// CHECK: define void @main() -float4 main(float2 uv : TEXCOORD) : SV_Target { - const Texture2D ltex = tex; - const SamplerState lsamp = samp; - const RWBuffer lbuf = buf; - const ByteAddressBuffer lbab = bab; - const StructuredBuffer lsb = sb; - - // CHECK: call %dx.types.ResRet.f32 @dx.op.sample.f32(i32 60, - float4 sampled = ltex.Sample(lsamp, uv); - // CHECK: call %dx.types.ResRet.f32 @dx.op.textureLoad.f32(i32 66, - float4 loaded = ltex.Load(int3(0, 0, 0)); - // CHECK: call %dx.types.ResRet.f32 @dx.op.bufferLoad.f32(i32 68, - float4 fromBuf = lbuf.Load(0); - // CHECK: call %dx.types.ResRet.i32 @dx.op.rawBufferLoad.i32(i32 139, {{.*}}, i32 0, i32 undef, - uint raw = lbab.Load(0); - // CHECK: call %dx.types.ResRet.i32 @dx.op.rawBufferLoad.i32(i32 139, {{.*}}, i32 0, i32 0, - int si = lsb.Load(0); - - // CHECK: call %dx.types.Dimensions @dx.op.getDimensions(i32 72, - uint w, h, l; - ltex.GetDimensions(0, w, h, l); - - // CHECK: call void @dx.op.bufferStore.f32(i32 69, - lbuf[1] = sampled; - - return sampled + loaded + fromBuf + float4(raw, si, w, h); -} diff --git a/tools/clang/test/HLSLFileCheck/hlsl/template/InstantiateObjectMethods.hlsl b/tools/clang/test/HLSLFileCheck/hlsl/template/InstantiateObjectMethods.hlsl index f9471a83a4..21f404bf48 100644 --- a/tools/clang/test/HLSLFileCheck/hlsl/template/InstantiateObjectMethods.hlsl +++ b/tools/clang/test/HLSLFileCheck/hlsl/template/InstantiateObjectMethods.hlsl @@ -18,7 +18,6 @@ float4 main() : SV_Target { // CHECK: CXXMemberCallExpr 0x{{[0-9a-fA-F]+}} 'vector' // CHECK-NEXT: MemberExpr 0x{{[0-9a-fA-F]+}} '' .Load -// CHECK-NEXT: ImplicitCastExpr 0x{{[0-9a-fA-F]+}} 'const Texture2D >' // CHECK-NEXT: CXXMemberCallExpr 0x{{[0-9a-fA-F]+}} 'Texture2D >':'Texture2D >' // CHECK-NEXT: MemberExpr 0x{{[0-9a-fA-F]+}} '' .Get // CHECK-NEXT: CXXThisExpr 0x{{[0-9a-fA-F]+}} 'MyTex2D diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugBreakInstrumentationInHelperFunction.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugBreakInstrumentationInHelperFunction.hlsl new file mode 100644 index 0000000000..3755f05285 --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugBreakInstrumentationInHelperFunction.hlsl @@ -0,0 +1,34 @@ +// RUN: %dxc -Emain -Tcs_6_10 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debugbreak-instrumentation -hlsl-dxilemit | %FileCheck %s + +// The debug-break pipeline shares the annotation prepass, so it also sees a +// module whose helpers are inlined away. A DebugBreak inside a [noinline] +// helper must still be found and instrumented once the helper is part of the +// entry point. + +// The helper does not appear as a separate function. +// CHECK-NOT: define {{.*}}BreakInHelper + +// The prepass reports no surviving helper, so PIX offers one steppable range. +// CHECK-NOT: UninlinedFunction: +// CHECK: InstructionRange: {{[0-9]+}} {{[0-9]+}} main cs +// CHECK-NOT: InstructionRange: + +// The break is still recorded, from inside the entry point. +// CHECK: %PixUAVHandle = call %dx.types.Handle @dx.op.createHandleFromBinding( +// CHECK: %DebugBreakBitSet = call i32 @dx.op.atomicBinOp.i32(i32 78, %dx.types.Handle +// CHECK-NOT: @dx.op.debugBreak + +RWStructuredBuffer Output : register(u0); + +[noinline] +uint BreakInHelper(uint value) +{ + DebugBreak(); + return value + 1; +} + +[numthreads(1, 1, 1)] +void main(uint3 threadId : SV_DispatchThreadID) +{ + Output[0] = BreakInHelper(threadId.x); +} diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugHullPatchConstantFunction.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugHullPatchConstantFunction.hlsl new file mode 100644 index 0000000000..dc39758be3 --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugHullPatchConstantFunction.hlsl @@ -0,0 +1,79 @@ +// RUN: %dxc -Emain -Ths_6_2 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=1,parameter1=2 -hlsl-dxilemit | %FileCheck %s + +// The runtime invokes a hull shader patch-constant function rather than the +// entry point does, so the inlining keeps it and the annotation pass numbers it +// as a steppable range of its own. An uninstrumented range emits no trace +// record, so a user who steps into the patch-constant body sees instructions +// with no values behind them. The pass instruments it as well as the entry +// point. +// +// SV_OutputControlPointID only means something in the control point phase, so +// the patch-constant function selects an invocation by primitive alone. + +// Two functions are numbered, so the helper that both of them call is inlined +// away. No third range is advertised, and none survives uninlined. +// CHECK-DAG: InstructionRange: {{[0-9]+ [0-9]+}} main hs +// CHECK-DAG: InstructionRange: {{[0-9]+ [0-9]+}} PatchConstantFunction +// CHECK-NOT: InstructionRange: +// CHECK-NOT: UninlinedFunction: + +// The patch-constant function selects on the primitive alone. +// CHECK: define void @"\01?PatchConstantFunction +// CHECK: %PrimId = call i32 @dx.op.primitiveID.i32(i32 108) +// CHECK-NEXT: %CompareToPrimId = icmp eq i32 %PrimId, 1 +// CHECK-NEXT: br i1 %CompareToPrimId, label %PIXInterestingBlock, label %PIXNonInterestingBlock + +// The entry point selects on the control point and the primitive. +// CHECK: define void @main() +// CHECK: %ControlPointId = call i32 @dx.op.outputControlPointID.i32(i32 107) +// CHECK-NEXT: %PrimId = call i32 @dx.op.primitiveID.i32(i32 108) +// CHECK-NEXT: %CompareToPrimId = icmp eq i32 %PrimId, 1 +// CHECK-NEXT: %CompareToControlPointId = icmp eq i32 %ControlPointId, 2 +// CHECK-NEXT: %CompareBoth = and i1 %CompareToControlPointId, %CompareToPrimId +// CHECK-NEXT: br i1 %CompareBoth, label %PIXInterestingBlock, label %PIXNonInterestingBlock + +// CHECK-NOT: HullHelper + +struct HsConstantData +{ + float Edges[3] : SV_TessFactor; + float Inside : SV_InsideTessFactor; +}; + +struct ControlPoint +{ + float3 position : WORLDPOS; +}; + +struct OutputPoint +{ + float3 vPosition : BEZIERPOS; +}; + +[noinline] +float HullHelper(float value) +{ + return value * 2.f; +} + +HsConstantData PatchConstantFunction(InputPatch ip) +{ + HsConstantData Output; + Output.Edges[0] = HullHelper(ip[0].position.x); + Output.Edges[1] = 8; + Output.Edges[2] = 8; + Output.Inside = 8; + return Output; +} + +[domain("tri")] +[partitioning("integer")] +[outputtopology("triangle_cw")] +[outputcontrolpoints(3)] +[patchconstantfunc("PatchConstantFunction")] +OutputPoint main(InputPatch ip, uint i : SV_OutputControlPointID) +{ + OutputPoint Output; + Output.vPosition = ip[i].position * HullHelper(2.f); + return Output; +} diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugNoInlineHelperFunction.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugNoInlineHelperFunction.hlsl new file mode 100644 index 0000000000..9da14081dc --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugNoInlineHelperFunction.hlsl @@ -0,0 +1,60 @@ +// RUN: %dxc -Emain -Tcs_6_2 /Od /Zi %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=1,parameter1=2,parameter2=3 -hlsl-dxilemit | %FileCheck %s +// RUN: %dxc -Emain -Tcs_6_2 /Od /Zi %s | %opt -S -dxil-dbg-value-to-dbg-declare -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=1,parameter1=2,parameter2=3 -hlsl-dxilemit | %FileCheck %s -check-prefixes=LOCALS,TRACED + +// PIX names a shader invocation by the stream of records that one thread writes +// into the debug UAV, and maps that stream to exactly one function. A helper +// instrumented as a function of its own writes its records under a second +// invocation identity for a thread that runs once, and PIX discards them. +// +// The passes inline a helper into the entry point before anything is numbered. +// PIX recovers the helper frame from the inlinedAt chain of each inlined +// instruction, and the helper locals stay attributed to the helper. + +// One function gives one invocation identity. +// CHECK: InstructionRange: {{[0-9]+}} {{[0-9]+}} main cs +// CHECK-NOT: InstructionRange: + +// Every helper is inlined, so the pass reports none as surviving. +// CHECK-NOT: UninlinedFunction: + +// CHECK: define void @main() +// CHECK-NOT: define {{.*}}ScaleHelper + +// Debug info still names the helper, so PIX rebuilds the call stack. +// CHECK: !DISubprogram(name: "ScaleHelper" +// CHECK: inlinedAt: + +// A helper local must be traced as well as scoped, or PIX reads it as +// unavailable. main holds no float local of its own, so a float alloca that +// carries a virtual register belongs to the inlined helper. +// TRACED: [[SCALED:%[0-9]+]] = alloca [1 x float], i32 0, !pix-alloca-reg + +// The helper local stays scoped to the helper, which puts it under the correct +// frame in the PIX locals view. +// LOCALS: call void @llvm.dbg.declare(metadata [1 x float]* [[SCALED]],{{.*}}; var:"scaled" + +// The store that traces the local carries no debug location. Requiring +// !pix-dxil-inst-num to follow the pointer operand immediately checks this. +// llvm::InlineFunction stamps the call site location onto each inlined +// instruction that carries none, so shadow storage must not exist yet when the +// helper is inlined. +// TRACED: [[SCALEDGEP:%[0-9]+]] = getelementptr [1 x float], [1 x float]* [[SCALED]], i32 0, i32 0 +// TRACED-NEXT: store float %{{[A-Za-z0-9_.]+}}, float* [[SCALEDGEP]], !pix-dxil-inst-num {{![0-9]+}}, !pix-alloca-reg-write + +// LOCALS: ![[HELPER:[0-9]+]] = !DISubprogram(name: "ScaleHelper" +// LOCALS: !DILocalVariable({{.*}}name: "scaled", scope: ![[HELPER]], + +RWStructuredBuffer Output : register(u0); + +[noinline] +float ScaleHelper(float value) +{ + float scaled = value * 3.f; + return scaled; +} + +[numthreads(1, 1, 1)] +void main(uint3 threadId : SV_DispatchThreadID) +{ + Output[threadId.x] = ScaleHelper(threadId.y); +} diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugStoreOpcodeByShaderModel.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugStoreOpcodeByShaderModel.hlsl new file mode 100644 index 0000000000..38652bd01f --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugStoreOpcodeByShaderModel.hlsl @@ -0,0 +1,21 @@ +// RUN: %dxc -Emain -Tcs_6_0 %s | %opt -S -hlsl-dxil-debug-instrumentation,UAVSize=1024 -hlsl-dxilemit | %FileCheck %s -check-prefix=SM60 +// RUN: %dxc -Emain -Tcs_6_1 %s | %opt -S -hlsl-dxil-debug-instrumentation,UAVSize=1024 -hlsl-dxilemit | %FileCheck %s -check-prefix=SM61 +// RUN: %dxc -Emain -Tcs_6_2 %s | %opt -S -hlsl-dxil-debug-instrumentation,UAVSize=1024 -hlsl-dxilemit | %FileCheck %s -check-prefix=SM62 + +// SM60: %PIX_DebugUAV_Handle = call %dx.types.Handle @dx.op.createHandle +// SM60-DAG: call void @dx.op.bufferStore.i32(i32 69, %dx.types.Handle %PIX_DebugUAV_Handle +// SM60-DAG: declare void @dx.op.bufferStore.i32(i32, %dx.types.Handle, i32, i32, i32, i32, i32, i32, i8) + +// SM61: %PIX_DebugUAV_Handle = call %dx.types.Handle @dx.op.createHandle +// SM61-DAG: call void @dx.op.bufferStore.i32(i32 69, %dx.types.Handle %PIX_DebugUAV_Handle +// SM61-DAG: declare void @dx.op.bufferStore.i32(i32, %dx.types.Handle, i32, i32, i32, i32, i32, i32, i8) + +// SM62: %PIX_DebugUAV_Handle = call %dx.types.Handle @dx.op.createHandle +// SM62-DAG: call void @dx.op.rawBufferStore.i32(i32 140, %dx.types.Handle %PIX_DebugUAV_Handle +// SM62-DAG: declare void @dx.op.rawBufferStore.i32(i32, %dx.types.Handle, i32, i32, i32, i32, i32, i32, i8, i32) + +[RootSignature("")] +[numthreads(1, 1, 1)] +void main(uint threadId : SV_DispatchThreadID) { + uint value = threadId; +} diff --git a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInHelperFunction.hlsl b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInHelperFunction.hlsl new file mode 100644 index 0000000000..bae3fcf097 --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInHelperFunction.hlsl @@ -0,0 +1,38 @@ +// RUN: %dxc -Emain -Tps_6_0 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-non-uniform-resource-index-instrumentation -hlsl-dxilemit | %FileCheck %s + +// The annotation prepass inlines away each non-entry function of a non-library +// module, because PIX cannot attribute a separately instrumented function to an +// invocation. See PIXPassHelpers::InlineNonEntryFunctions. The +// non-uniform-resource-index pipeline shares that prepass for the instruction +// ordinals its diagnostics use, so an unqualified dynamic index inside a +// [noinline] helper must still be diagnosed once the helper is part of the +// entry point. +// +// The [noinline] attribute is necessary. Without it the front end inlines the +// helper and the prepass has nothing to do. The helper signature holds scalars +// only, because a function that reaches DXIL in a non-library module takes and +// returns no vector. + +// The helper does not appear as a separate function. +// CHECK-NOT: define {{.*}}IndexInHelper + +// The dynamic index is still reported, and it is addressed to a real ordinal +// instead of to bit 0. +// CHECK: @dx.op.waveActiveAllEqual +// CHECK: shl i32 %{{[0-9]+}}, {{[1-9][0-9]*}} +// CHECK: @dx.op.atomicBinOp.i32(i32 78 +// CHECK-NOT: NuriNotInstrumentedMissingInstructionNumber + +Texture2D tex[8] : register(t0); + +[noinline] +float IndexInHelper(float u, float v) +{ + uint index = u * v; + return tex[index].Load(int3(0, 0, 0)).x; +} + +float4 main(float2 uv : TEXCOORD0) : SV_TARGET +{ + return IndexInHelper(uv.x, uv.y); +} diff --git a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInstructionNumber.hlsl b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInstructionNumber.hlsl index ab79b41af4..d002fed467 100644 --- a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInstructionNumber.hlsl +++ b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexInstructionNumber.hlsl @@ -2,17 +2,14 @@ // With the annotation prepass in place, the diagnostic is addressed to // the ordinal of the createHandle that performed the unmarked dynamic -// indexing. The pass encodes the ordinal as a shift within a 32-bit word -// (InstructionNumber % 32), with InstructionNumber / 32 selecting the -// word. A zero shift addresses bit 0 in that word; ordinals 32, 64, etc. -// also have a zero shift in later words. +// indexing. The pass encodes that ordinal as a shift. A shift of zero +// aliases the diagnostic onto bit 0. // // Match any non-zero shift rather than a literal ordinal. A createHandle // whose index comes from an interpolated input is never the first // numbered instruction. // CHECK-NOT: NuriNotInstrumentedMissingInstructionNumber -// CHECK: FoundDynamicIndexingNoNuri // CHECK: @dx.op.waveActiveAllEqual // CHECK: shl i32 %{{[0-9]+}}, {{[1-9][0-9]*}} // CHECK: @dx.op.atomicBinOp.i32(i32 78 diff --git a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexLibraryHelper.hlsl b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexLibraryHelper.hlsl index cf0a16883b..348385b083 100644 --- a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexLibraryHelper.hlsl +++ b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexLibraryHelper.hlsl @@ -5,7 +5,6 @@ // a non-zero instruction ordinal. // CHECK-NOT: NuriNotInstrumentedMissingInstructionNumber -// CHECK: FoundDynamicIndexingNoNuri // CHECK: define void {{.*}}IndexInHelper // CHECK: @dx.op.waveActiveAllEqual // CHECK: shl i32 %{{[0-9]+}}, {{[1-9][0-9]*}} diff --git a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexNoInstructionNumbers.hlsl b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexNoInstructionNumbers.hlsl index 7053fbdefc..d810401b4e 100644 --- a/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexNoInstructionNumbers.hlsl +++ b/tools/clang/test/HLSLFileCheck/pix/NonUniformResourceIndexNoInstructionNumbers.hlsl @@ -10,7 +10,6 @@ // CHECK-NOT: FoundDynamicIndexingNoNuri // CHECK: NuriNotInstrumentedMissingInstructionNumber -// CHECK-NOT: !"PixUAVResource" // CHECK-NOT: @dx.op.waveActiveAllEqual // CHECK-NOT: @dx.op.atomicBinOp diff --git a/tools/clang/test/SemaHLSL/extern-redeclarations.hlsl b/tools/clang/test/SemaHLSL/extern-redeclarations.hlsl deleted file mode 100644 index e69ce190ff..0000000000 --- a/tools/clang/test/SemaHLSL/extern-redeclarations.hlsl +++ /dev/null @@ -1,19 +0,0 @@ -// RUN: %dxc -T lib_6_3 -verify %s - -extern float externThenDefinition; // expected-note {{previous definition is here}} -float externThenDefinition; // expected-error {{redefinition of 'externThenDefinition'}} - -float definitionThenExtern; // expected-note {{previous definition is here}} -extern float definitionThenExtern; // expected-error {{redefinition of 'definitionThenExtern'}} - -extern float duplicateExtern; // expected-note {{previous definition is here}} -extern float duplicateExtern; // expected-error {{redefinition of 'duplicateExtern'}} - -extern Texture2D externTextureThenDefinition; // expected-note {{previous definition is here}} -Texture2D externTextureThenDefinition; // expected-error {{redefinition of 'externTextureThenDefinition'}} - -Texture2D textureDefinitionThenExtern; // expected-note {{previous definition is here}} -extern Texture2D textureDefinitionThenExtern; // expected-error {{redefinition of 'textureDefinitionThenExtern'}} - -extern float mismatched; // expected-note {{previous declaration is here}} -int mismatched; // expected-error {{redefinition of 'mismatched' with a different type}} diff --git a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x.hlsl b/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x.hlsl deleted file mode 100644 index 14489255dc..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x.hlsl +++ /dev/null @@ -1,58 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 202x -ast-dump %s | FileCheck %s - -// Verify the parser accepts `const` instance methods in HLSL 202x and that -// overload resolution selects the const overload for const objects and the -// non-const overload for non-const objects, including for out-of-line -// definitions and class templates. - -struct S { - int x; - int get() { return 100; } - int get() const { return 200; } - int outOfLine() const; -}; - -// Two distinct overloads: one const-qualified, one not. -// CHECK: CXXMethodDecl [[NC:0x[0-9a-f]+]] {{.*}} used get 'int ()' -// CHECK: CXXMethodDecl [[C:0x[0-9a-f]+]] {{.*}} used get 'int () const' -// CHECK: CXXMethodDecl [[OOLDecl:0x[0-9a-f]+]] {{.*}} outOfLine 'int () const' - -int S::outOfLine() const { return x; } -// CHECK: CXXMethodDecl {{0x[0-9a-f]+}} parent {{0x[0-9a-f]+}} prev [[OOLDecl]] {{.*}} used outOfLine 'int () const' - -template struct W { - T v; - T get() { return v; } - T get() const { return v; } -}; - -// CHECK: ClassTemplateSpecializationDecl {{.*}} struct W definition -// CHECK: CXXMethodDecl [[WNC:0x[0-9a-f]+]] {{.*}} used get 'int ()' -// CHECK: CXXMethodDecl [[WC:0x[0-9a-f]+]] {{.*}} used get 'int () const' - -cbuffer CB { - S cs; // cs is const because it lives in a cbuffer. -}; - -float4 main() : SV_Target { - S s = {1}; - int a = s.get(); // expect non-const overload - int b = cs.get(); // expect const overload - int c = cs.outOfLine(); - W w = {2}; - const W cw = {3}; - int d = w.get(); // expect non-const overload - int e = cw.get(); // expect const overload - return float4(a, b, c, d + e); -} - -// CHECK: MemberExpr {{.*}} .get [[NC]] -// CHECK-NEXT: DeclRefExpr {{.*}} 'S' lvalue Var {{0x[0-9a-f]+}} 's' 'S' -// CHECK: MemberExpr {{.*}} .get [[C]] -// CHECK-NEXT: DeclRefExpr {{.*}} 'const S' lvalue Var {{0x[0-9a-f]+}} 'cs' 'const S' -// CHECK: MemberExpr {{.*}} .outOfLine -// CHECK-NEXT: DeclRefExpr {{.*}} 'const S' lvalue Var {{0x[0-9a-f]+}} 'cs' 'const S' -// CHECK: MemberExpr {{.*}} .get [[WNC]] -// CHECK-NEXT: DeclRefExpr {{.*}} 'W':'W' lvalue Var {{0x[0-9a-f]+}} 'w' -// CHECK: MemberExpr {{.*}} .get [[WC]] -// CHECK-NEXT: DeclRefExpr {{.*}} 'const W':'const W' lvalue Var {{0x[0-9a-f]+}} 'cw' diff --git a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x_errors.hlsl b/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x_errors.hlsl deleted file mode 100644 index 5149a530cc..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x_errors.hlsl +++ /dev/null @@ -1,63 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 202x -verify %s - -// Verify that const-correctness is enforced for HLSL 202x: a non-const -// instance method cannot be called on a const object, regardless of whether -// the const-ness comes from a cbuffer member, a ConstantBuffer, the -// implicit global cbuffer, an explicit `const` local, or `this` inside a -// const method. - -struct S { - int x; - int get() const { return x; } - int getNC() { return x; } // expected-note 6 {{'getNC' declared here}} - - int callNC() const { - return getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const S', but function is not marked const}} - } - int callNCThis() const { - return this.getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const S', but function is not marked const}} - } - - int dup() const const { return x; } // expected-warning {{duplicate 'const' declaration specifier}} - - static int staticConst() const { return 0; } // expected-error {{static member function cannot have 'const' qualifier}} -}; - -template struct W { - T v; - T get() const { return v; } - T getNC() { return v; } // expected-note {{'getNC' declared here}} -}; - -int freeConst() const { return 0; } // expected-error {{non-member function cannot have 'const' qualifier}} - -// `const` is only accepted on function declarators, not in array bounds. -static int Arr[const 4]; // expected-error {{expected expression}} - -cbuffer CB { - S cs; -}; - -ConstantBuffer cb; - -S g; // implicit global cbuffer member - implicitly const. - -float4 main() : SV_Target { - // OK: const method on each kind of const object. - int a = cs.get(); - int b = cb.get(); - int c = g.get(); - const S ls = {1}; - int d = ls.get(); - const W lw = {2}; - int j = lw.get(); - - // Error: non-const method on const object. - int e = cs.getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const S', but function is not marked const}} - int f = cb.getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const S', but function is not marked const}} - int h = g.getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const S', but function is not marked const}} - int i = ls.getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const S', but function is not marked const}} - int k = lw.getNC(); // expected-error {{member function 'getNC' not viable: 'this' argument has type 'const W', but function is not marked const}} - - return float4(a, b, c, d) + float4(e, f, h, i) + j + k; -} diff --git a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x_this_const.hlsl b/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x_this_const.hlsl deleted file mode 100644 index c659487aa5..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_202x_this_const.hlsl +++ /dev/null @@ -1,33 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 202x -verify %s - -// Verify that inside a const-qualified instance method, 'this' refers to a -// const object and the object's fields cannot be modified. A non-const -// sibling method should still be able to mutate the same fields. - -struct S { - int x; - int arr[4]; - - void modify(int v) const { // expected-note 2 {{member function 'S::modify' is declared const here}} - x = v; // expected-error {{cannot assign to non-static data member within const member function 'modify'}} - arr[0] = v; // expected-error {{read-only variable is not assignable}} - x += v; // expected-error {{cannot assign to non-static data member within const member function 'modify'}} - } - - void mutate(int v) { - // Non-const method: mutation is fine. - x = v; - arr[0] = v; - } - - int read() const { - // Reading 'this' fields from a const method is fine. - return x + arr[0]; - } -}; - -float4 main() : SV_Target { - S s = {1, {2, 3, 4, 5}}; - s.mutate(7); - return s.read(); -} diff --git a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_pre202x.hlsl b/tools/clang/test/SemaHLSL/hlsl/classes/const_method_pre202x.hlsl deleted file mode 100644 index 92af86d549..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/classes/const_method_pre202x.hlsl +++ /dev/null @@ -1,20 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 2021 -verify %s - -// Verify that pre-HLSL 202x rejects `const`-qualified instance methods, and -// that non-const methods remain callable on const (cbuffer) objects for -// backwards compatibility. - -struct S { - int x; - int get() const { return x; } // expected-error {{const-qualified member functions are unsupported in HLSL before 202x}} - int getNC() { return x; } -}; - -cbuffer CB { - S cs; -}; - -float4 main() : SV_Target { - S s = {1}; - return s.get() + cs.getNC(); -} diff --git a/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type-ast.hlsl b/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type-ast.hlsl index 547912aee4..05241a9b8d 100644 --- a/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type-ast.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type-ast.hlsl @@ -7,8 +7,8 @@ struct S { __builtin_LinAlgMatrix handle; }; -// CHECK: VarDecl {{.*}} global_handle '__builtin_LinAlgMatrix':'__builtin_LinAlgMatrix' static -static __builtin_LinAlgMatrix global_handle; +// CHECK: VarDecl {{.*}} global_handle '__builtin_LinAlgMatrix':'__builtin_LinAlgMatrix' +__builtin_LinAlgMatrix global_handle; // CHECK: FunctionDecl {{.*}} f1 'void (__builtin_LinAlgMatrix)' // CHECK: ParmVarDecl {{.*}} m '__builtin_LinAlgMatrix':'__builtin_LinAlgMatrix' diff --git a/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type.hlsl b/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type.hlsl index 858254cede..c9c885f3ec 100644 --- a/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/linalg/builtins/builtin-matrix-handle-type.hlsl @@ -1,22 +1,15 @@ // REQUIRES: dxil-1-10 // RUN: %dxc -T lib_6_10 -verify %s -// expected-error@+1 {{global variable 'global_handle' containing a linear algebra matrix must be declared 'static'}} __builtin_LinAlgMatrix global_handle; static __builtin_LinAlgMatrix static_handle; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in groupshared variables}} groupshared __builtin_LinAlgMatrix gs_handle; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in groupshared variables}} -static groupshared __builtin_LinAlgMatrix static_gs_handle; - -// expected-error@+1 {{global variable 'array' containing a linear algebra matrix must be declared 'static'}} __builtin_LinAlgMatrix array[2]; cbuffer CB { - // expected-error@+1 {{global variable 'cb_handle' containing a linear algebra matrix must be declared 'static'}} __builtin_LinAlgMatrix cb_handle; }; @@ -24,7 +17,7 @@ struct S { __builtin_LinAlgMatrix handle; }; -static S s; +S s; void f1(__builtin_LinAlgMatrix m); diff --git a/tools/clang/test/SemaHLSL/hlsl/linalg/interpreted-vector-type-errors.hlsl b/tools/clang/test/SemaHLSL/hlsl/linalg/interpreted-vector-type-errors.hlsl deleted file mode 100644 index cc7837a93c..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/linalg/interpreted-vector-type-errors.hlsl +++ /dev/null @@ -1,72 +0,0 @@ -// REQUIRES: dxil-1-10 -// RUN: %dxc -T lib_6_10 -verify %s - -#include -using namespace dx::linalg; - -using MatrixTy = - Matrix; - -void validTypes(float4 FloatVec, uint4 UintVec, - vector PackedVec, - vector SignedPackedVec) { - MakeInterpretedVector(FloatVec); - MakeInterpretedVector(UintVec); - MakeInterpretedVector(PackedVec); - MakeInterpretedVector(SignedPackedVec); - - Convert(FloatVec); - Convert(UintVec); - Convert(PackedVec); - Convert(FloatVec); - Convert(PackedVec); -} - -void invalidFactories(float4 FloatVec, uint4 UintVec) { - // expected-error@+1{{no matching function for call to 'MakeInterpretedVector'}} - MakeInterpretedVector(UintVec); - // expected-error@+1{{no matching function for call to 'MakeInterpretedVector'}} - MakeInterpretedVector(FloatVec); - // expected-error@+1{{no matching function for call to 'MakeInterpretedVector'}} - MakeInterpretedVector(UintVec); - // expected-error@+1{{no matching function for call to 'MakeInterpretedVector'}} - MakeInterpretedVector(UintVec); - // expected-error@+1{{no matching function for call to 'MakeInterpretedVector'}} - MakeInterpretedVector(UintVec); - - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(UintVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(FloatVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(UintVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(FloatVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(UintVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(UintVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(UintVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(FloatVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(FloatVec); - // expected-error@+1{{no matching function for call to 'Convert'}} - Convert(FloatVec); -} - -void invalidConsumers(MatrixTy Mat, ByteAddressBuffer Buf, float4 Bias) { - InterpretedVector Invalid = {0}; - - // expected-error@+1{{no matching function for call to 'Multiply'}} - Multiply(Mat, Invalid); - // expected-error@+1{{no matching function for call to 'MultiplyAdd'}} - MultiplyAdd(Mat, Invalid, Bias); - - VectorRef Ref = {Buf, 0}; - // expected-error@+1{{no matching function for call to 'MultiplyAdd'}} - MultiplyAdd(Mat, Invalid, Ref); -} - -// expected-note@dx/linalg.h:* 39{{candidate template ignored}} diff --git a/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-global-error.hlsl b/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-global-error.hlsl deleted file mode 100644 index dfc43efef2..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-global-error.hlsl +++ /dev/null @@ -1,102 +0,0 @@ -// REQUIRES: dxil-1-10 -// RUN: %dxc -T cs_6_10 -E main -verify %s - -// Globals containing LinAlg matrices are mutable state, not constant buffer -// data, so they must be explicitly declared 'static'. - -#include -using namespace dx::linalg; - -using MatrixTy = - Matrix; - -using HandleTy = __builtin_LinAlgMatrix - [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, - MatrixUse::Accumulator, MatrixScope::Wave)]]; - -struct MatrixState { - MatrixTy Mat; -}; - -struct DerivedState : MatrixState { - float F; -}; - -struct NumericState { - float4 V; -}; - -template struct Wrapper { - T Val; -}; - -// expected-error@+1 {{global variable 'GMat' containing a linear algebra matrix must be declared 'static'}} -MatrixTy GMat; -// expected-error@+1 {{global variable 'GMatArr' containing a linear algebra matrix must be declared 'static'}} -MatrixTy GMatArr[2]; -// expected-error@+1 {{global variable 'GMatArr2D' containing a linear algebra matrix must be declared 'static'}} -MatrixTy GMatArr2D[2][3]; -// expected-error@+1 {{global variable 'GState' containing a linear algebra matrix must be declared 'static'}} -MatrixState GState; -// expected-error@+1 {{global variable 'GDerived' containing a linear algebra matrix must be declared 'static'}} -DerivedState GDerived; -// expected-error@+1 {{global variable 'GWrapped' containing a linear algebra matrix must be declared 'static'}} -Wrapper GWrapped; -// expected-error@+1 {{global variable 'GHandle' containing a linear algebra matrix must be declared 'static'}} -HandleTy GHandle; -// expected-error@+1 {{global variable 'GRawHandle' containing a linear algebra matrix must be declared 'static'}} -__builtin_LinAlgMatrix GRawHandle; - -// LinAlg matrices cannot be stored in groupshared memory, even when static. -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::Accumulator, MatrixScope::Wave)]]') is not allowed in groupshared variables}} -groupshared MatrixTy GSMat; -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::Accumulator, MatrixScope::Wave)]]') is not allowed in groupshared variables}} -static groupshared MatrixTy SGSMatArr[2]; -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::Accumulator, MatrixScope::Wave)]]') is not allowed in groupshared variables}} -groupshared DerivedState GSDerived; -// expected-note@dx/linalg.h:* 3 {{field declared here}} -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::Accumulator, MatrixScope::Wave)]]') is not allowed in groupshared variables}} -static groupshared HandleTy SGSHandle; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in groupshared variables}} -groupshared __builtin_LinAlgMatrix GSRawHandle; -groupshared float4 GSNumeric; - -cbuffer CB { - // expected-error@+1 {{global variable 'CBMat' containing a linear algebra matrix must be declared 'static'}} - MatrixTy CBMat; - static MatrixTy CBStaticMat; -}; - -namespace NS { -// expected-error@+1 {{global variable 'NSMat' containing a linear algebra matrix must be declared 'static'}} -MatrixTy NSMat; -static MatrixTy NSStaticMat; -} - -static MatrixTy SMat; -static MatrixTy SMatArr[2]; -static MatrixState SState; -static DerivedState SDerived; -static Wrapper SWrapped; -static HandleTy SHandle; -static __builtin_LinAlgMatrix SRawHandle; - -// Globals without LinAlg matrices are unaffected. -NumericState GNumeric; -Wrapper GWrappedFloat; -RWByteAddressBuffer Output; - -struct StaticMember { - static MatrixTy Mat; -}; - -typedef MatrixTy MatrixTypedef; - -[numthreads(1, 1, 1)] -void main() { - MatrixTy LocalMat = MatrixTy::Splat(1.0f); - static MatrixTy LocalStaticMat; - SMat = LocalMat; - Output.Store(0, SMat.Get(0) + GNumeric.V.x + GWrappedFloat.Val); -} diff --git a/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-resource-error.hlsl b/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-resource-error.hlsl deleted file mode 100644 index 8e0a1f00ce..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-resource-error.hlsl +++ /dev/null @@ -1,133 +0,0 @@ -// REQUIRES: dxil-1-10 -// RUN: %dxc -T lib_6_10 -Wno-hlsl-availability -verify %s - -// LinAlg matrices are opaque, thread-local values and cannot be stored in any -// resource, node record, patch, or stream, either directly or nested in a -// struct. - -#include -using namespace dx::linalg; - -using MatrixTy = - Matrix; - -using HandleTy = __builtin_LinAlgMatrix - [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, - MatrixScope::Wave)]]; - -// The Matrix class holds its handle in a private field. -// expected-note@dx/linalg.h:* 9 {{field declared here}} - -struct MatrixState { - MatrixTy Mat; -}; - -struct HandleState { - HandleTy Handle; // expected-note 6 {{field declared here}} -}; - -struct RawHandleState { - __builtin_LinAlgMatrix Handle; // expected-note 6 {{field declared here}} -}; - -struct DerivedState : MatrixState { - float F; -}; - -struct Numeric { - float4 V; -}; - -// ConstantBuffer and TextureBuffer. -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer CBMat; -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer CBState; -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer CBDerived; -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer CBHandle; -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer CBHandleState; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer<__builtin_LinAlgMatrix> CBRawHandle; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in ConstantBuffers or TextureBuffers}} -ConstantBuffer CBRawHandleState; -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -TextureBuffer TBState; -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in ConstantBuffers or TextureBuffers}} -TextureBuffer TBHandleState; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in ConstantBuffers or TextureBuffers}} -TextureBuffer TBRawHandleState; - -// Structured buffers. -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in structured buffers}} -StructuredBuffer SBMat; -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in structured buffers}} -RWStructuredBuffer RWSBState; -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in structured buffers}} -AppendStructuredBuffer ASBHandleState; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in structured buffers}} -ConsumeStructuredBuffer CSBRawHandleState; -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in structured buffers}} -RasterizerOrderedStructuredBuffer<__builtin_LinAlgMatrix> ROSBRawHandle; - -// Typed buffers and textures. -// expected-error@+1 {{'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') cannot be used as a type parameter}} -Buffer BufHandle; -// expected-error@+1 {{'__builtin_LinAlgMatrix' cannot be used as a type parameter}} -RWBuffer<__builtin_LinAlgMatrix> RWBufRawHandle; -// expected-error@+2 {{'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') cannot be used as a type parameter}} -// expected-note@+1 {{usage of 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') found in field '__handle' of type 'MatrixTy' (aka 'Matrix')}} -Texture2D TexMat; -// expected-error@+2 {{'__builtin_LinAlgMatrix' cannot be used as a type parameter}} -// expected-note@+1 {{usage of '__builtin_LinAlgMatrix' found in field 'Handle' of type 'RawHandleState'}} -RWTexture3D RWTexRawHandleState; - -// Numeric element types remain valid. -ConstantBuffer CBNumeric; -TextureBuffer TBNumeric; -StructuredBuffer SBNumeric; -Buffer BufNumeric; - -ByteAddressBuffer BAB; -RWByteAddressBuffer RWBAB; - -void ByteAddressBufferTemplates() { - // expected-error@+2 {{Explicit template arguments on intrinsic Load must be a single numeric type}} - // expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in builtin template parameters}} - MatrixState A = BAB.Load(0); - // expected-error@+2 {{Explicit template arguments on intrinsic Load must be a single numeric type}} - // expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in builtin template parameters}} - __builtin_LinAlgMatrix B = BAB.Load<__builtin_LinAlgMatrix>(0); - RawHandleState C; - // expected-error@+2 {{Explicit template arguments on intrinsic Store must be a single numeric type}} - // expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in builtin template parameters}} - RWBAB.Store(0, C); - HandleState D; - // expected-error@+2 {{Explicit template arguments on intrinsic Store must be a single numeric type}} - // expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in builtin template parameters}} - RWBAB.Store(0, D); - - Numeric N = BAB.Load(0); - RWBAB.Store(0, N); -} - -// Node records. -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in node records}} -void NodeInputMat(DispatchNodeInputRecord In); -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in node records}} -void NodeOutputHandle(NodeOutput Out); -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in node records}} -void NodeOutputRawHandle(NodeOutput Out); -void NodeNumeric(DispatchNodeInputRecord In, NodeOutput Out); - -// Tessellation patches and geometry streams. -// expected-error@+1 {{object 'HandleT' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in tessellation patches}} -void PatchMat(InputPatch P); -// expected-error@+1 {{object 'HandleTy' (aka '__builtin_LinAlgMatrix [[__LinAlgMatrix_Attributes(ComponentType::F32, 4, 4, MatrixUse::A, MatrixScope::Wave)]]') is not allowed in tessellation patches}} -void PatchHandle(OutputPatch P); -// expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in geometry streams}} -void StreamRawHandle(inout PointStream S); -void PatchAndStreamNumeric(InputPatch P, - inout TriangleStream S); diff --git a/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-shader-interface-error.hlsl b/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-shader-interface-error.hlsl deleted file mode 100644 index c3d7811c5c..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/linalg/linalg-matrix-shader-interface-error.hlsl +++ /dev/null @@ -1,57 +0,0 @@ -// REQUIRES: dxil-1-10 -// RUN: %dxc -T lib_6_10 -Wno-hlsl-availability -verify %s - -// LinAlg matrices are opaque, thread-local values and cannot cross shader -// interfaces such as entry signatures, ray payloads, or hit attributes. - -struct RawHandleState { - __builtin_LinAlgMatrix Handle; // expected-note 6 {{field declared here}} -}; - -struct Numeric { - float4 V; -}; - -// expected-error@+2 {{object '__builtin_LinAlgMatrix' is not allowed in entry function parameters}} -[shader("compute")] [numthreads(1, 1, 1)] -void CSParam(RawHandleState S : A) {} - -// expected-error@+2 {{object '__builtin_LinAlgMatrix' is not allowed in entry function return type}} -[shader("pixel")] -RawHandleState PSReturn() : SV_Target { - RawHandleState S; - return S; -} - -RaytracingAccelerationStructure AS; - -[shader("raygeneration")] -void RGTrace() { - RawHandleState Payload; - RayDesc Ray = (RayDesc)0; - // expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in user-defined struct parameter}} - TraceRay(AS, 0, 0xff, 0, 1, 0, Ray, Payload); - - Numeric NumericPayload; - TraceRay(AS, 0, 0xff, 0, 1, 0, Ray, NumericPayload); -} - -// expected-error@+3 {{object '__builtin_LinAlgMatrix' is not allowed in entry function parameters}} -// expected-error@+2 {{payload parameter 'Payload' must be a user-defined type composed of only numeric types}} -[shader("closesthit")] -void CHPayload(inout RawHandleState Payload, - BuiltInTriangleIntersectionAttributes Attrs) {} - -[shader("intersection")] -void ISReportHit() { - RawHandleState Attrs; - // expected-error@+1 {{object '__builtin_LinAlgMatrix' is not allowed in attributes}} - ReportHit(0.0, 0, Attrs); -} - -// expected-error@+2 {{object '__builtin_LinAlgMatrix' is not allowed in entry function parameters}} -[shader("compute")] [numthreads(1, 1, 1)] -void CSArrayParam(RawHandleState S[2] : B) {} - -// Non-entry functions may take and return LinAlg matrices. -RawHandleState Passthrough(RawHandleState S) { return S; } diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/const_hitobject_set_errors.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/const_hitobject_set_errors.hlsl deleted file mode 100644 index 4b895f328f..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/const_hitobject_set_errors.hlsl +++ /dev/null @@ -1,22 +0,0 @@ -// RUN: %dxc -T lib_6_9 -HV 202x -verify %s - -// SetShaderTableIndex mutates the dx::HitObject, so it must not be callable -// on a const-qualified instance. Const accessors should still work. - -void use_const(const dx::HitObject ho) { - // expected-error@+2{{no matching member function for call to 'SetShaderTableIndex'}} - // expected-note@+1 1+ {{but method is not marked const}} - ho.SetShaderTableIndex(1); - - // Const accessors are fine. - bool isMiss = ho.IsMiss(); - bool isHit = ho.IsHit(); - uint idx = ho.GetShaderTableIndex(); - (void)isMiss; (void)isHit; (void)idx; -} - -[shader("raygeneration")] -void main() { - dx::HitObject ho; - use_const(ho); -} diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_accessors.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_accessors.hlsl index 92c988bb6a..841ab69090 100644 --- a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_accessors.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_accessors.hlsl @@ -5,7 +5,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetHitKind // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetHitKind 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetHitKind 'unsigned int () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetHitKind 'unsigned int ()' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 366 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -13,7 +13,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetInstanceID // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetInstanceID 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetInstanceID 'unsigned int () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetInstanceID 'unsigned int ()' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 367 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -21,7 +21,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetInstanceIndex // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetInstanceIndex 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetInstanceIndex 'unsigned int () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetInstanceIndex 'unsigned int ()' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 368 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -29,7 +29,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetObjectRayDirection // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetObjectRayDirection 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectRayDirection 'vector () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectRayDirection 'vector ()' extern // AST-NEXT: | | | |-TemplateArgument type 'vector':'vector' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 369 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -37,7 +37,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetObjectRayOrigin // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetObjectRayOrigin 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectRayOrigin 'vector () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectRayOrigin 'vector ()' extern // AST-NEXT: | | | |-TemplateArgument type 'vector':'vector' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 370 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -45,7 +45,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetObjectToWorld3x4 // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetObjectToWorld3x4 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectToWorld3x4 'matrix () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectToWorld3x4 'matrix ()' extern // AST-NEXT: | | | |-TemplateArgument type 'matrix':'matrix' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 371 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -53,7 +53,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetObjectToWorld4x3 // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetObjectToWorld4x3 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectToWorld4x3 'matrix () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetObjectToWorld4x3 'matrix ()' extern // AST-NEXT: | | | |-TemplateArgument type 'matrix':'matrix' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 372 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -61,7 +61,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetPrimitiveIndex // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetPrimitiveIndex 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetPrimitiveIndex 'unsigned int () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetPrimitiveIndex 'unsigned int ()' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 373 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -69,7 +69,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetRayFlags // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetRayFlags 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetRayFlags 'unsigned int () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetRayFlags 'unsigned int ()' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 374 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -77,7 +77,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetRayTCurrent // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetRayTCurrent 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetRayTCurrent 'float () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetRayTCurrent 'float ()' extern // AST-NEXT: | | | |-TemplateArgument type 'float' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 375 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -85,7 +85,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetRayTMin // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetRayTMin 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetRayTMin 'float () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetRayTMin 'float ()' extern // AST-NEXT: | | | |-TemplateArgument type 'float' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 376 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -93,7 +93,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetShaderTableIndex // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetShaderTableIndex 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetShaderTableIndex 'unsigned int () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetShaderTableIndex 'unsigned int ()' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 377 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -101,7 +101,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetWorldRayDirection // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetWorldRayDirection 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldRayDirection 'vector () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldRayDirection 'vector ()' extern // AST-NEXT: | | | |-TemplateArgument type 'vector':'vector' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 378 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -109,7 +109,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetWorldRayOrigin // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetWorldRayOrigin 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldRayOrigin 'vector () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldRayOrigin 'vector ()' extern // AST-NEXT: | | | |-TemplateArgument type 'vector':'vector' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 379 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -117,7 +117,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetWorldToObject3x4 // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetWorldToObject3x4 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldToObject3x4 'matrix () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldToObject3x4 'matrix ()' extern // AST-NEXT: | | | |-TemplateArgument type 'matrix':'matrix' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 380 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -125,7 +125,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> GetWorldToObject4x3 // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetWorldToObject4x3 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldToObject4x3 'matrix () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetWorldToObject4x3 'matrix ()' extern // AST-NEXT: | | | |-TemplateArgument type 'matrix':'matrix' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 381 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -133,7 +133,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> IsHit // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit IsHit 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used IsHit 'bool () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used IsHit 'bool ()' extern // AST-NEXT: | | | |-TemplateArgument type 'bool' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 383 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -141,7 +141,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> IsMiss // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit IsMiss 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used IsMiss 'bool () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used IsMiss 'bool ()' extern // AST-NEXT: | | | |-TemplateArgument type 'bool' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 384 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -149,7 +149,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> IsNop // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit IsNop 'TResult () const' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used IsNop 'bool () const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used IsNop 'bool ()' extern // AST-NEXT: | | | |-TemplateArgument type 'bool' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 385 // AST-NEXT: | | | |-ConstAttr {{[^ ]+}} <> Implicit @@ -159,7 +159,7 @@ // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TRootConstantOffsetInBytes // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit LoadLocalRootTableConstant 'TResult (TRootConstantOffsetInBytes) const' // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> RootConstantOffsetInBytes 'TRootConstantOffsetInBytes' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used LoadLocalRootTableConstant 'unsigned int (unsigned int) const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used LoadLocalRootTableConstant 'unsigned int (unsigned int)' extern // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-TemplateArgument type 'unsigned int' // AST-NEXT: | | | |-ParmVarDecl {{[^ ]+}} <> LoadLocalRootTableConstant 'unsigned int' @@ -169,7 +169,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> SetShaderTableIndex // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TRecordIndex -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit SetShaderTableIndex 'TResult (TRecordIndex)' +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit SetShaderTableIndex 'TResult (TRecordIndex) const' // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> RecordIndex 'TRecordIndex' // AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used SetShaderTableIndex 'void (unsigned int)' extern // AST-NEXT: | | | |-TemplateArgument type 'void' diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_attributes.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_attributes.hlsl index 1466edfe47..6d0291b691 100644 --- a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_attributes.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_attributes.hlsl @@ -8,7 +8,7 @@ // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TAttributes // AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit GetAttributes 'TResult (TAttributes &) const' // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> Attributes 'TAttributes &' -// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetAttributes 'void (CustomAttrs &) const' extern +// AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used GetAttributes 'void (CustomAttrs &)' extern // AST-NEXT: | | | |-TemplateArgument type 'void' // AST-NEXT: | | | |-TemplateArgument type 'CustomAttrs' // AST-NEXT: | | | |-ParmVarDecl {{[^ ]+}} <> GetAttributes 'CustomAttrs &&__restrict' diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_fromrayquery.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_fromrayquery.hlsl index b1b9ec9c81..c07f6ee5f5 100644 --- a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_fromrayquery.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_fromrayquery.hlsl @@ -5,7 +5,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> FromRayQuery // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class Trq -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit FromRayQuery 'TResult (Trq)' static +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit FromRayQuery 'TResult (Trq) const' static // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> rq 'Trq' // AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used FromRayQuery 'dx::HitObject (RayQuery)' static // AST-NEXT: | | | |-TemplateArgument type 'dx::HitObject' @@ -19,7 +19,7 @@ // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class Trq // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class THitKind // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TAttributes -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit FromRayQuery 'TResult (Trq, THitKind, TAttributes)' static +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit FromRayQuery 'TResult (Trq, THitKind, TAttributes) const' static // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> rq 'Trq' // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> HitKind 'THitKind' // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> Attributes 'TAttributes' diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_make_ast.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_make_ast.hlsl index 5bb485867f..ab604b9213 100644 --- a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_make_ast.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_make_ast.hlsl @@ -14,7 +14,7 @@ // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TRayFlags // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TMissShaderIndex // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TRay -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit MakeMiss 'TResult (TRayFlags, TMissShaderIndex, TRay)' static +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit MakeMiss 'TResult (TRayFlags, TMissShaderIndex, TRay) const' static // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> RayFlags 'TRayFlags' // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> MissShaderIndex 'TMissShaderIndex' // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> Ray 'TRay' @@ -31,7 +31,7 @@ // AST: | | |-FunctionTemplateDecl {{[^ ]+}} <> MakeNop // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit MakeNop 'TResult ()' static +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit MakeNop 'TResult () const' static // AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used MakeNop 'dx::HitObject ()' static // AST-NEXT: | | | |-TemplateArgument type 'dx::HitObject' // AST-NEXT: | | | |-HLSLIntrinsicAttr {{[^ ]+}} <> Implicit "op" "" 358 diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_traceinvoke_ast.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_traceinvoke_ast.hlsl index 23c0979e3d..d979c7301f 100644 --- a/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_traceinvoke_ast.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/objects/HitObject/hitobject_traceinvoke_ast.hlsl @@ -4,7 +4,7 @@ // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TResult // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class Tho // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TPayload -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit Invoke 'TResult (Tho, TPayload &)' static +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit Invoke 'TResult (Tho, TPayload &) const' static // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> ho 'Tho' // AST-NEXT: | | | | `-ParmVarDecl {{[^ ]+}} <> Payload 'TPayload &' // AST-NEXT: | | | `-CXXMethodDecl {{[^ ]+}} <> used Invoke 'void (dx::HitObject, Payload &)' static @@ -26,7 +26,7 @@ // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TMissShaderIndex // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TRay // AST-NEXT: | | | |-TemplateTypeParmDecl {{[^ ]+}} <> class TPayload -// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit TraceRay 'TResult (TAccelerationStructure, TRayFlags, TInstanceInclusionMask, TRayContributionToHitGroupIndex, TMultiplierForGeometryContributionToHitGroupIndex, TMissShaderIndex, TRay, TPayload &)' static +// AST-NEXT: | | | |-CXXMethodDecl {{[^ ]+}} <> implicit TraceRay 'TResult (TAccelerationStructure, TRayFlags, TInstanceInclusionMask, TRayContributionToHitGroupIndex, TMultiplierForGeometryContributionToHitGroupIndex, TMissShaderIndex, TRay, TPayload &) const' static // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> AccelerationStructure 'TAccelerationStructure' // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> RayFlags 'TRayFlags' // AST-NEXT: | | | | |-ParmVarDecl {{[^ ]+}} <> InstanceInclusionMask 'TInstanceInclusionMask' diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/const_rayquery_mutator_errors.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/const_rayquery_mutator_errors.hlsl deleted file mode 100644 index c267d2c535..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/objects/const_rayquery_mutator_errors.hlsl +++ /dev/null @@ -1,38 +0,0 @@ -// RUN: %dxc -T lib_6_5 -HV 202x -verify %s - -// Verify that RayQuery mutators (TraceRayInline, Proceed, Abort, and the -// Commit* helpers) cannot be invoked on a const-qualified RayQuery, while -// the read-only accessors (CommittedStatus, CandidateType, ...) remain -// callable. - -RaytracingAccelerationStructure RTAS : register(t0); - -void use_const(const RayQuery cq) { - RayDesc desc = (RayDesc)0; - // expected-error@+2{{no matching member function for call to 'TraceRayInline'}} - // expected-note@+1 1+ {{but method is not marked const}} - cq.TraceRayInline(RTAS, RAY_FLAG_NONE, 0xff, desc); - // expected-error@+2{{no matching member function for call to 'Proceed'}} - // expected-note@+1 1+ {{but method is not marked const}} - bool ok = cq.Proceed(); - // expected-error@+2{{no matching member function for call to 'Abort'}} - // expected-note@+1 1+ {{but method is not marked const}} - cq.Abort(); - // expected-error@+2{{no matching member function for call to 'CommitNonOpaqueTriangleHit'}} - // expected-note@+1 1+ {{but method is not marked const}} - cq.CommitNonOpaqueTriangleHit(); - // expected-error@+2{{no matching member function for call to 'CommitProceduralPrimitiveHit'}} - // expected-note@+1 1+ {{but method is not marked const}} - cq.CommitProceduralPrimitiveHit(1.0f); - - // Const accessors are fine. - uint s = cq.CommittedStatus(); - uint t = cq.CandidateType(); - (void)s; (void)t; -} - -[shader("raygeneration")] -void main() { - RayQuery q; - use_const(q); -} diff --git a/tools/clang/test/SemaHLSL/hlsl/objects/const_resource_reassign_errors.hlsl b/tools/clang/test/SemaHLSL/hlsl/objects/const_resource_reassign_errors.hlsl deleted file mode 100644 index d58ab2f8e1..0000000000 --- a/tools/clang/test/SemaHLSL/hlsl/objects/const_resource_reassign_errors.hlsl +++ /dev/null @@ -1,24 +0,0 @@ -// RUN: %dxc -T ps_6_0 -E main -HV 202x -verify %s - -// Verify that const-declared local resource objects cannot be reassigned. -// All built-in HLSL resource/handle types should reject assignment when -// declared const, but their (const) instance methods must remain callable. - -Texture2D tex : register(t0); -Texture2D tex2 : register(t1); -SamplerState samp : register(s0); -RWBuffer buf : register(u0); -RWBuffer buf2 : register(u1); - -float4 main(float2 uv : TEXCOORD) : SV_Target { - const Texture2D ltex = tex; // expected-note{{variable 'ltex' declared const here}} - const RWBuffer lbuf = buf; // expected-note{{variable 'lbuf' declared const here}} - - // Reassigning a const resource is an error. - ltex = tex2; // expected-error{{cannot assign to variable 'ltex' with const-qualified type 'const Texture2D'}} - lbuf = buf2; // expected-error{{cannot assign to variable 'lbuf' with const-qualified type 'const RWBuffer'}} - - // But const methods on the handle are fine. - float4 v = ltex.Sample(samp, uv); - return v + lbuf.Load(0); -} diff --git a/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-EmptyNodeOutputArrayTypes.hlsl b/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-EmptyNodeOutputArrayTypes.hlsl index bae8918ffe..94c481247d 100644 --- a/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-EmptyNodeOutputArrayTypes.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-EmptyNodeOutputArrayTypes.hlsl @@ -29,7 +29,7 @@ void node_2_0( // CHECK-NEXT:| | |-TemplateTypeParmDecl 0x{{.+}} <> class Tcount // CHECK-NEXT:| | |-CXXMethodDecl 0x{{.+}} <> implicit GroupIncrementOutputCount 'TResult (Tcount) const' // CHECK-NEXT:| | | `-ParmVarDecl 0x{{.+}} <> count 'Tcount' -// CHECK-NEXT:| | `-CXXMethodDecl 0x[[GroupIncrementOutputCount:[0-9a-f]+]] <> used GroupIncrementOutputCount 'void (unsigned int) const' extern +// CHECK-NEXT:| | `-CXXMethodDecl 0x[[GroupIncrementOutputCount:[0-9a-f]+]] <> used GroupIncrementOutputCount 'void (unsigned int)' extern // CHECK-NEXT:| | |-TemplateArgument type 'void' // CHECK-NEXT:| | |-TemplateArgument type 'unsigned int' // CHECK-NEXT:| | |-ParmVarDecl 0x{{.+}} <> GroupIncrementOutputCount 'unsigned int' @@ -62,13 +62,12 @@ void node_2_0( // CHECK-NEXT: |-CompoundStmt 0x{{.+}} // CHECK-NEXT: | `-CXXMemberCallExpr 0x{{.+}} 'void' // CHECK-NEXT: | |-MemberExpr 0x{{.+}} '' .GroupIncrementOutputCount 0x[[GroupIncrementOutputCount]] -// CHECK-NEXT: | | `-ImplicitCastExpr 0x{{.+}} 'const EmptyNodeOutput' -// CHECK-NEXT: | | `-CXXOperatorCallExpr 0x{{.+}} 'EmptyNodeOutput' -// CHECK-NEXT: | | |-ImplicitCastExpr 0x{{.+}} 'EmptyNodeOutput (*)(unsigned int)' -// CHECK-NEXT: | | | `-DeclRefExpr 0x{{.+}} 'EmptyNodeOutput (unsigned int)' lvalue CXXMethod 0x[[SUB]] 'operator[]' 'EmptyNodeOutput (unsigned int)' -// CHECK-NEXT: | | |-DeclRefExpr 0x{{.+}} 'EmptyNodeOutputArray' lvalue ParmVar 0x[[Param]] 'OutputArray_2_0' 'EmptyNodeOutputArray' -// CHECK-NEXT: | | `-ImplicitCastExpr 0x{{.+}} 'unsigned int' -// CHECK-NEXT: | | `-IntegerLiteral 0x{{.+}}{{.+}} 'literal int' 1 +// CHECK-NEXT: | | `-CXXOperatorCallExpr 0x{{.+}} 'EmptyNodeOutput' +// CHECK-NEXT: | | |-ImplicitCastExpr 0x{{.+}} 'EmptyNodeOutput (*)(unsigned int)' +// CHECK-NEXT: | | | `-DeclRefExpr 0x{{.+}} 'EmptyNodeOutput (unsigned int)' lvalue CXXMethod 0x[[SUB]] 'operator[]' 'EmptyNodeOutput (unsigned int)' +// CHECK-NEXT: | | |-DeclRefExpr 0x{{.+}} 'EmptyNodeOutputArray' lvalue ParmVar 0x[[Param]] 'OutputArray_2_0' 'EmptyNodeOutputArray' +// CHECK-NEXT: | | `-ImplicitCastExpr 0x{{.+}} 'unsigned int' +// CHECK-NEXT: | | `-IntegerLiteral 0x{{.+}}{{.+}} 'literal int' 1 // CHECK-NEXT: | `-ImplicitCastExpr 0x{{.+}} 'unsigned int' // CHECK-NEXT: | `-IntegerLiteral 0x{{.+}} 'literal int' 10 // CHECK-NEXT: |-HLSLNumThreadsAttr 0x{{.+}} 1 1 1 diff --git a/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-NodeOutputArrayTypes.hlsl b/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-NodeOutputArrayTypes.hlsl index 56582c32e5..52b558a07a 100644 --- a/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-NodeOutputArrayTypes.hlsl +++ b/tools/clang/test/SemaHLSL/hlsl/workgraph/ast-NodeOutputArrayTypes.hlsl @@ -83,7 +83,7 @@ void node_1_1( // CHECK-NEXT:| |-FunctionTemplateDecl 0x{{.+}} <> OutputComplete // CHECK-NEXT:| | |-TemplateTypeParmDecl 0x{{.+}} <> class TResult // CHECK-NEXT:| | |-CXXMethodDecl 0x{{.+}} <> OutputComplete 'TResult () const' -// CHECK-NEXT:| | `-CXXMethodDecl 0x[[OutComplete:[0-9a-f]+]] <> used OutputComplete 'void () const' extern +// CHECK-NEXT:| | `-CXXMethodDecl 0x[[OutComplete:[0-9a-f]+]] <> used OutputComplete 'void ()' extern // CHECK-NEXT:| | |-TemplateArgument type 'void' // CHECK-NEXT:| | `-HLSLIntrinsicAttr 0x{{.+}} <> Implicit "op" "" {{[0-9]+}} // CHECK-NEXT:| `-CXXDestructorDecl 0x{{.+}} <> implicit referenced ~ThreadNodeOutputRecords 'void () noexcept' inline @@ -124,7 +124,7 @@ void node_1_1( // CHECK-NEXT:| | |-TemplateTypeParmDecl 0x{{.+}} <> class TnumRecords // CHECK-NEXT:| | |-CXXMethodDecl 0x{{.+}} <> GetThreadNodeOutputRecords 'TResult (TnumRecords) const' // CHECK-NEXT:| | | `-ParmVarDecl 0x{{.+}} <> numRecords 'TnumRecords' -// CHECK-NEXT:| | `-CXXMethodDecl 0x[[GetThreadNodeOutputRecords:[0-9a-f]+]] <> used GetThreadNodeOutputRecords 'ThreadNodeOutputRecords (unsigned int) const' extern +// CHECK-NEXT:| | `-CXXMethodDecl 0x[[GetThreadNodeOutputRecords:[0-9a-f]+]] <> used GetThreadNodeOutputRecords 'ThreadNodeOutputRecords (unsigned int)' extern // CHECK-NEXT:| | |-TemplateArgument type 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' // CHECK-NEXT:| | |-TemplateArgument type 'unsigned int' // CHECK-NEXT:| | |-ParmVarDecl 0x{{.+}} <> GetThreadNodeOutputRecords 'unsigned int' @@ -165,19 +165,17 @@ void node_1_1( // CHECK-NEXT: | | `-VarDecl 0x[[OutRec:[0-9a-f]+]] col:36 used outRec 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' cinit // CHECK-NEXT: | | `-CXXMemberCallExpr 0x{{.+}} 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' // CHECK-NEXT: | | |-MemberExpr 0x{{.+}} '' .GetThreadNodeOutputRecords 0x[[GetThreadNodeOutputRecords]] -// CHECK-NEXT: | | | `-ImplicitCastExpr 0x{{.+}} 'const NodeOutput' -// CHECK-NEXT: | | | `-CXXOperatorCallExpr 0x{{.+}} 'NodeOutput':'NodeOutput' -// CHECK-NEXT: | | | |-ImplicitCastExpr 0x{{.+}} 'NodeOutput (*)(unsigned int)' -// CHECK-NEXT: | | | | `-DeclRefExpr 0x{{.+}} 'NodeOutput (unsigned int)' lvalue CXXMethod 0x[[SUB]] 'operator[]' 'NodeOutput (unsigned int)' -// CHECK-NEXT: | | | |-DeclRefExpr 0x{{.+}} 'NodeOutputArray':'NodeOutputArray' lvalue ParmVar 0x[[ParmVar]] 'OutputArray_1_1' 'NodeOutputArray':'NodeOutputArray' -// CHECK-NEXT: | | | `-ImplicitCastExpr 0x{{.+}} 'unsigned int' -// CHECK-NEXT: | | | `-IntegerLiteral 0x{{.+}} 'literal int' 1 +// CHECK-NEXT: | | | `-CXXOperatorCallExpr 0x{{.+}} 'NodeOutput':'NodeOutput' +// CHECK-NEXT: | | | |-ImplicitCastExpr 0x{{.+}} 'NodeOutput (*)(unsigned int)' +// CHECK-NEXT: | | | | `-DeclRefExpr 0x{{.+}} 'NodeOutput (unsigned int)' lvalue CXXMethod 0x[[SUB]] 'operator[]' 'NodeOutput (unsigned int)' +// CHECK-NEXT: | | | |-DeclRefExpr 0x{{.+}} 'NodeOutputArray':'NodeOutputArray' lvalue ParmVar 0x[[ParmVar]] 'OutputArray_1_1' 'NodeOutputArray':'NodeOutputArray' +// CHECK-NEXT: | | | `-ImplicitCastExpr 0x{{.+}} 'unsigned int' +// CHECK-NEXT: | | | `-IntegerLiteral 0x{{.+}} 'literal int' 1 // CHECK-NEXT: | | `-ImplicitCastExpr 0x{{.+}} 'unsigned int' // CHECK-NEXT: | | `-IntegerLiteral 0x{{.+}} 'literal int' 2 // CHECK-NEXT: | `-CXXMemberCallExpr 0x{{.+}} 'void' // CHECK-NEXT: | `-MemberExpr 0x{{.+}} '' .OutputComplete 0x[[OutComplete]] -// CHECK-NEXT: | `-ImplicitCastExpr 0x{{.+}} 'const ThreadNodeOutputRecords' lvalue -// CHECK-NEXT: | `-DeclRefExpr 0x{{.+}} 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' lvalue Var 0x[[OutRec]] 'outRec' 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' +// CHECK-NEXT: | `-DeclRefExpr 0x{{.+}} 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' lvalue Var 0x[[OutRec]] 'outRec' 'ThreadNodeOutputRecords':'ThreadNodeOutputRecords' // CHECK-NEXT: |-HLSLNumThreadsAttr 0x{{.+}} 1 1 1 // CHECK-NEXT: |-HLSLNodeDispatchGridAttr 0x{{.+}} 1 1 1 // CHECK-NEXT: |-HLSLNodeLaunchAttr 0x{{.+}} "broadcasting" diff --git a/tools/clang/test/SemaHLSL/v2021-static-assert-not-keyword.hlsl b/tools/clang/test/SemaHLSL/v2021-static-assert-not-keyword.hlsl deleted file mode 100644 index 1f00248eab..0000000000 --- a/tools/clang/test/SemaHLSL/v2021-static-assert-not-keyword.hlsl +++ /dev/null @@ -1,11 +0,0 @@ -// RUN: %dxc -T lib_6_3 -HV 2021 -verify %s - -// In HLSL versions earlier than 202x, 'static_assert' remains available as an -// ordinary identifier. -// expected-no-diagnostics - -int static_assert = 1; - -int use_static_assert(int static_assert) { - return static_assert; -} diff --git a/tools/clang/test/SemaHLSL/v202x/static-assert.hlsl b/tools/clang/test/SemaHLSL/v202x/static-assert.hlsl deleted file mode 100644 index 43f65fab0c..0000000000 --- a/tools/clang/test/SemaHLSL/v202x/static-assert.hlsl +++ /dev/null @@ -1,49 +0,0 @@ -// RUN: %dxc -T lib_6_3 -HV 202x -verify %s - -// In HLSL 202x, 'static_assert' is recognized as a keyword and accepts -// both the C++11 form (condition + message) and the C++17 form -// (condition only). Statically-verifiable, true conditions produce no -// diagnostic; false conditions produce an error; non-constant -// conditions produce an error indicating the expression is not a -// constant. - -// --- True conditions: no diagnostics expected for these. --- - -static_assert(1 == 1, "math: equality is reflexive"); -static_assert(sizeof(float) == 4, "float is 32 bits"); -static_assert(2 + 2 == 4, "arithmetic still works"); - -// C++17 form: no message. -static_assert(1 < 2); -static_assert(sizeof(int) == 4); - -// In a function body. -void in_function_body() { - static_assert(sizeof(float4) == 16, "float4 is 16 bytes"); - static_assert(sizeof(float4) == 16); -} - -// In a struct/class member-specification. -struct S { - static_assert(sizeof(float) == 4, "float is 4 bytes inside a struct"); - static_assert(sizeof(float) == 4); - float member; -}; - -// --- False conditions: each should trigger a static_assert failure. --- - -static_assert(1 == 2, "one is not two"); // expected-error{{static_assert failed "one is not two"}} -static_assert(sizeof(float) == 8, "float is not 8 bytes"); // expected-error{{static_assert failed "float is not 8 bytes"}} - -// C++17 form (no message) failing. -static_assert(1 == 2); // expected-error{{static_assert failed}} - -// --- Non-constant conditions: should be rejected as not constant. --- - -cbuffer CB { int g_runtime; }; - -static_assert(g_runtime == 0, "runtime value"); // expected-error{{static_assert expression is not an integral constant expression}} - -void uses_param(int p) { // expected-note{{declared here}} - static_assert(p == 0, "parameter is runtime"); // expected-error{{static_assert expression is not an integral constant expression}} expected-note{{read of non-const variable 'p' is not allowed in a constant expression}} -} diff --git a/tools/clang/unittests/HLSL/CompilerTest.cpp b/tools/clang/unittests/HLSL/CompilerTest.cpp index 048773af19..f590a8745c 100644 --- a/tools/clang/unittests/HLSL/CompilerTest.cpp +++ b/tools/clang/unittests/HLSL/CompilerTest.cpp @@ -4364,34 +4364,7 @@ void VerifyDivByZeroThrows() { VERIFY_IS_TRUE(bCaughtExpectedException); } -static bool IsX64EmulatedOnArm64() { -#if defined(_M_X64) - using IsWow64Process2Fn = BOOL(WINAPI *)(HANDLE, USHORT *, USHORT *); - HMODULE kernel32 = GetModuleHandleW(L"kernel32.dll"); - IsWow64Process2Fn isWow64Process2 = reinterpret_cast( - GetProcAddress(kernel32, "IsWow64Process2")); - if (!isWow64Process2) - return false; - - USHORT processMachine = IMAGE_FILE_MACHINE_UNKNOWN; - USHORT nativeMachine = IMAGE_FILE_MACHINE_UNKNOWN; - return isWow64Process2(GetCurrentProcess(), &processMachine, - &nativeMachine) && - nativeMachine == IMAGE_FILE_MACHINE_ARM64; -#else - return false; -#endif -} - TEST_F(CompilerTest, CodeGenFloatingPointEnvironment) { - if (IsX64EmulatedOnArm64()) { - WEX::Logging::Log::Comment( - L"Skipping floating-point environment test under x64 emulation on " - L"ARM64."); - WEX::Logging::Log::Result(WEX::Logging::TestResults::Skipped); - return; - } - unsigned int fpOriginal; VERIFY_IS_TRUE(_controlfp_s(&fpOriginal, 0, 0) == 0); diff --git a/tools/clang/unittests/HLSL/PixTest.cpp b/tools/clang/unittests/HLSL/PixTest.cpp index c7c6556881..631c064d6f 100644 --- a/tools/clang/unittests/HLSL/PixTest.cpp +++ b/tools/clang/unittests/HLSL/PixTest.cpp @@ -16,7 +16,6 @@ #include #include #include -#include #include #include #include @@ -170,7 +169,11 @@ class PixTest : public ::testing::Test { TEST_METHOD(ToolsUav_ExtendsEveryGlobalRootSignatureSubobject) TEST_METHOD(ToolsUav_PreservesGlobalRootSignatureSourceText) TEST_METHOD(DebugInstrumentation_RawBufferShaderFlagDeclared) + TEST_METHOD(DebugInstrumentation_NothingInstrumentableAddsNoUav) TEST_METHOD(ToolsUav_RootSignatureSerializationFailurePreservesSignature) + TEST_METHOD(DebugInstrumentation_SM60UsesBufferStore) + TEST_METHOD(DebugInstrumentation_SM61UsesBufferStore) + TEST_METHOD(DebugInstrumentation_SM62UsesRawBufferStore) TEST_METHOD(ToolsUav_ExtendingRootSignaturePreservesUnrelatedParameterFlags) TEST_METHOD(ConstantColor_UnusedIntOverloadIsErased) TEST_METHOD(ConstantColor_NoTargetOverloadsAreErased) @@ -184,7 +187,7 @@ class PixTest : public ::testing::Test { TEST_METHOD(DxilPIXDXRInvocationsLog_SanityTest) TEST_METHOD(DxilPIXDXRInvocationsLog_EmbeddedRootSigs) - TEST_METHOD(DxilPIXDXRInvocationsLog_ZeroCapacityStillCountsInvocations) + TEST_METHOD(DxilPIXDXRInvocationsLog_ZeroCapacityEmitsNothing) TEST_METHOD(DxilPIXDXRInvocationsLog_OneEntryUsesEntryCountBound) TEST_METHOD(DxilPIXDXRInvocationsLog_ExactCapacityUsesEntryCountBound) TEST_METHOD(DxilPIXDXRInvocationsLog_OverflowGuardValidates) @@ -219,6 +222,15 @@ class PixTest : public ::testing::Test { TEST_METHOD(NonUniformResourceIndex_DescriptorHeap) TEST_METHOD(NonUniformResourceIndex_Raytracing) + TEST_METHOD(HelperInlining_NoEntryFunctionLeavesModuleAlone) + TEST_METHOD(HelperInlining_SurvivingHelperIsReportedByExactName) + TEST_METHOD(HelperInlining_SurvivingHelperIsReportedAfterEarlierPrepass) + TEST_METHOD(HelperInlining_InlinedHelperIsNotReported) + TEST_METHOD(HelperInlining_InlinedHelperValidates) + TEST_METHOD(HelperInlining_HullPatchConstantFunctionValidates) + TEST_METHOD(HelperInlining_NonUniformResourceIndexHelperValidates) + TEST_METHOD(HelperInlining_DebugBreakHelperValidates) + // Control tests for the PIX pass validation harness below // (validateInstrumentedModule / verifyInstrumentedModuleIsValid). TEST_METHOD(Validation_ControlValidModulePasses) @@ -374,6 +386,19 @@ class PixTest : public ::testing::Test { // Runs the virtual-register annotation pass over textual IR and returns the // pass report. Textual IR builds a module shape that HLSL does not express. std::vector RunAnnotationPassOnText(const std::string &irText) { + return RunPassesOnText(irText, {L"-dxil-annotate-with-virtual-regs"}); + } + + // Runs both prepasses that inline, in the order the debug pipeline uses, so + // the annotation pass sees a module another pass already inlined. + std::vector + RunDbgValueAndAnnotationPassesOnText(const std::string &irText) { + return RunPassesOnText(irText, {L"-dxil-dbg-value-to-dbg-declare", + L"-dxil-annotate-with-virtual-regs"}); + } + + std::vector RunPassesOnText(const std::string &irText, + std::vector passes) { CComPtr pSource; CreateBlobFromText(m_dllSupport, irText.c_str(), &pSource); @@ -383,7 +408,7 @@ class PixTest : public ::testing::Test { std::vector Options; Options.push_back(L"-S"); Options.push_back(L"-opt-mod-passes"); - Options.push_back(L"-dxil-annotate-with-virtual-regs"); + Options.insert(Options.end(), passes.begin(), passes.end()); CComPtr pOptimizedModule; CComPtr pText; @@ -393,6 +418,30 @@ class PixTest : public ::testing::Test { return Tokenize(BlobToUtf8(pText).c_str(), "\n"); } + static bool AnyLineContains(const std::vector &lines, + const char *needle) { + for (const std::string &line : lines) { + if (line.find(needle) != std::string::npos) { + return true; + } + } + return false; + } + + // Collects the whole payload of each report record with the given prefix, so + // a test pins the exact record rather than a substring of it. + static std::vector + RecordsWithPrefix(const std::vector &lines, + const std::string &prefix) { + std::vector records; + for (const std::string &line : lines) { + if (line.compare(0, prefix.size(), prefix) == 0) { + records.push_back(line.substr(prefix.size())); + } + } + return records; + } + // Replaces the one occurrence of needle, and fails the test when the text // does not hold exactly one. static std::string ReplaceOnlyOccurrence(const std::string &text, @@ -407,6 +456,77 @@ class PixTest : public ::testing::Test { return result; } + // Returns the module with a global holding the helper's address, which makes + // the helper reachable other than by a direct call and so unreachable for the + // inliner. HLSL itself expresses no such module. + static std::string TakeAddressOfHelper(const std::string &disassembly, + const char *mangledHelperName) { + const std::string entryDefinition = "define void @main() {"; + return ReplaceOnlyOccurrence( + disassembly, entryDefinition, + "@PIXTestHelperAddress = internal global float (float)* @\"\\01?" + + std::string(mangledHelperName) + "\"\n\n" + entryDefinition); + } + + // Counts the definitions in a disassembly whose name holds the given text. + static unsigned CountFunctionDefinitions(const std::string &disassembly, + const char *nameFragment) { + unsigned count = 0; + for (const std::string &line : Tokenize(disassembly.c_str(), "\n")) { + if (line.compare(0, strlen("define "), "define ") == 0 && + line.find(nameFragment) != std::string::npos) { + count++; + } + } + return count; + } + + static unsigned CountOccurrences(const std::string &text, + const char *needle) { + unsigned count = 0; + size_t needleLength = strlen(needle); + for (size_t position = text.find(needle); position != std::string::npos; + position = text.find(needle, position + needleLength)) { + count++; + } + return count; + } + + // Counts calls to a named operation. A disassembly also holds one declare + // line per operation, which is not a call. + static unsigned CountCallsTo(const std::string &disassembly, + const char *operation) { + unsigned count = 0; + for (const std::string &line : Tokenize(disassembly.c_str(), "\n")) { + if (line.find(operation) != std::string::npos && + line.find("call ") != std::string::npos) { + count++; + } + } + return count; + } + + // Returns the text of the one definition whose name holds the given fragment, + // so a test asserts about one function rather than the whole module. + static std::string ExtractFunctionBody(const std::string &disassembly, + const char *nameFragment) { + std::string body; + bool inside = false; + for (const std::string &line : Tokenize(disassembly.c_str(), "\n")) { + if (!inside) { + inside = line.compare(0, strlen("define "), "define ") == 0 && + line.find(nameFragment) != std::string::npos; + } else if (line.compare(0, 1, "}") == 0) { + break; + } + if (inside) { + body += line + "\n"; + } + } + VERIFY_IS_FALSE(body.empty()); + return body; + } + SinglePassOutput runSinglePass(IDxcBlob *Dxil, LPCWSTR PassOption) { CComPtr Optimizer; VERIFY_SUCCEEDED( @@ -908,16 +1028,11 @@ static constexpr uint32_t ToolsRegisterSpace = static_cast(-2); static bool HasDxrInvocationLogEntryCountCheck(std::vector const &lines, unsigned expectedEntryCount) { - const std::string comparison = - "icmp ult i32 %EntryIndexResult, " + std::to_string(expectedEntryCount); - for (const std::string &Line : lines) { - const size_t Position = Line.find(comparison); - if (Position != std::string::npos) { - const size_t End = Position + comparison.size(); - if (End == Line.size() || - !std::isdigit(static_cast(Line[End]))) { - return true; - } + const std::string expectedSuffix = ", " + std::to_string(expectedEntryCount); + for (auto const &line : lines) { + if (line.find("icmp ult i32 %EntryIndexResult") != std::string::npos && + line.find(expectedSuffix) != std::string::npos) { + return true; } } return false; @@ -3935,6 +4050,22 @@ void main(uint threadId : SV_DispatchThreadID) "debug instrumentation shader flags"); } +TEST_F(PixTest, DebugInstrumentation_NothingInstrumentableAddsNoUav) { + const char *source = R"x( +export float Helper(float x) +{ + return x * 2; +} +)x"; + + auto compiled = Compile(m_dllSupport, source, L"lib_6_6", {L"-Od"}); + CComPtr dxil = FindModule(DFCC_ShaderDebugInfoDXIL, compiled); + auto output = RunDebugPass(dxil); + + VERIFY_ARE_EQUAL( + 0, countToolsUAVRecords(Tokenize(Disassemble(output.blob), "\n"))); +} + TEST_F(PixTest, ToolsUav_RootSignatureSerializationFailurePreservesSignature) { const char *Source = R"x( [numthreads(1, 1, 1)] @@ -4013,6 +4144,83 @@ void main() VERIFY_IS_TRUE(FoundRootSignature); } +static const char *kDebugStoreOpcodeComputeShader = R"x( +[numthreads(1, 1, 1)] +void main(uint threadId : SV_DispatchThreadID) +{ + uint value = threadId; +} +)x"; + +// RawBufferStore is illegal below shader model 6.2. Debug instrumentation of +// 6.0 and 6.1 shaders must emit BufferStore (opcode 69) instead, and the +// instrumented module must validate. The declaration and opcode are checked so +// a test cannot pass by skipping instrumentation. +static void VerifyDebugStoreOpcodeDisassembly(const std::string &disassembly, + bool expectRawBufferStore) { + VERIFY_IS_TRUE(disassembly.find("PIX_DebugUAV_Handle") != std::string::npos); + VERIFY_ARE_NOT_EQUAL( + 0u, PixTest::CountCallsTo(disassembly, "@dx.op.atomicBinOp.i32")); + + if (expectRawBufferStore) { + VERIFY_ARE_NOT_EQUAL( + 0u, PixTest::CountCallsTo(disassembly, "@dx.op.rawBufferStore.i32")); + VERIFY_IS_TRUE( + disassembly.find("call void @dx.op.rawBufferStore.i32(i32 " + "140, %dx.types.Handle %PIX_DebugUAV_Handle") != + std::string::npos); + VERIFY_IS_TRUE( + disassembly.find("declare void @dx.op.rawBufferStore.i32(i32, " + "%dx.types.Handle, i32, i32, i32, i32, i32, i32, i8, " + "i32)") != std::string::npos); + VERIFY_ARE_EQUAL(0u, + PixTest::CountCallsTo(disassembly, "@dx.op.bufferStore")); + VERIFY_IS_TRUE(disassembly.find("declare void @dx.op.bufferStore") == + std::string::npos); + } else { + VERIFY_ARE_NOT_EQUAL( + 0u, PixTest::CountCallsTo(disassembly, "@dx.op.bufferStore.i32")); + VERIFY_IS_TRUE(disassembly.find("call void @dx.op.bufferStore.i32(i32 69, " + "%dx.types.Handle %PIX_DebugUAV_Handle") != + std::string::npos); + VERIFY_IS_TRUE( + disassembly.find( + "declare void @dx.op.bufferStore.i32(i32, %dx.types.Handle, i32, " + "i32, i32, i32, i32, i32, i8)") != std::string::npos); + VERIFY_ARE_EQUAL( + 0u, PixTest::CountCallsTo(disassembly, "@dx.op.rawBufferStore")); + VERIFY_IS_TRUE(disassembly.find("declare void @dx.op.rawBufferStore") == + std::string::npos); + } +} + +TEST_F(PixTest, DebugInstrumentation_SM60UsesBufferStore) { + auto compiled = Compile(m_dllSupport, kDebugStoreOpcodeComputeShader, + L"cs_6_0", {L"-Od"}); + auto output = RunDebugPass(compiled); + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of an SM 6.0 compute shader"); + VerifyDebugStoreOpcodeDisassembly(Disassemble(output.blob), false); +} + +TEST_F(PixTest, DebugInstrumentation_SM61UsesBufferStore) { + auto compiled = Compile(m_dllSupport, kDebugStoreOpcodeComputeShader, + L"cs_6_1", {L"-Od"}); + auto output = RunDebugPass(compiled); + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of an SM 6.1 compute shader"); + VerifyDebugStoreOpcodeDisassembly(Disassemble(output.blob), false); +} + +TEST_F(PixTest, DebugInstrumentation_SM62UsesRawBufferStore) { + auto compiled = Compile(m_dllSupport, kDebugStoreOpcodeComputeShader, + L"cs_6_2", {L"-Od"}); + auto output = RunDebugPass(compiled); + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of an SM 6.2 compute shader"); + VerifyDebugStoreOpcodeDisassembly(Disassemble(output.blob), true); +} + TEST_F(PixTest, ToolsUav_ExtendingRootSignaturePreservesUnrelatedParameterFlags) { const char *Source = R"x( @@ -4471,51 +4679,45 @@ void MyMiss(inout MyPayload payload) RunDxilPIXDXRInvocationsLog(compiledLib); } -TEST_F(PixTest, DxilPIXDXRInvocationsLog_ZeroCapacityStillCountsInvocations) { - CComPtr CompiledLib = +TEST_F(PixTest, DxilPIXDXRInvocationsLog_ZeroCapacityEmitsNothing) { + auto compiledLib = Compile(m_dllSupport, kSingleMissInvocationLogShader, L"lib_6_6", {}); - CComPtr ZeroEntryOutput = - RunDxilPIXDXRInvocationsLog(CompiledLib, 0); - const std::string ZeroEntryDisassembly = Disassemble(ZeroEntryOutput); - const std::vector ZeroEntryLines = - Tokenize(ZeroEntryDisassembly, "\n"); + auto oneEntryOutput = RunDxilPIXDXRInvocationsLog(compiledLib, 1); + auto oneEntryLines = Tokenize(Disassemble(oneEntryOutput), "\n"); + VERIFY_ARE_EQUAL(2, countToolsUAVRecords(oneEntryLines)); - VERIFY_ARE_EQUAL(1, countToolsUAVRecords(ZeroEntryLines)); - VERIFY_IS_TRUE(ZeroEntryDisassembly.find("@dx.op.atomicBinOp.i32") != - std::string::npos); - VERIFY_IS_TRUE(ZeroEntryDisassembly.find("call void @dx.op.bufferStore.") == - std::string::npos); + auto zeroEntryOutput = RunDxilPIXDXRInvocationsLog(compiledLib, 0); + auto zeroEntryLines = Tokenize(Disassemble(zeroEntryOutput), "\n"); + VERIFY_ARE_EQUAL(0, countToolsUAVRecords(zeroEntryLines)); } TEST_F(PixTest, DxilPIXDXRInvocationsLog_OneEntryUsesEntryCountBound) { - CComPtr CompiledLib = + auto compiledLib = Compile(m_dllSupport, kSingleMissInvocationLogShader, L"lib_6_6", {}); - CComPtr Output = RunDxilPIXDXRInvocationsLog(CompiledLib, 1); - const std::vector Lines = Tokenize(Disassemble(Output), "\n"); + auto output = RunDxilPIXDXRInvocationsLog(compiledLib, 1); + auto lines = Tokenize(Disassemble(output), "\n"); - VERIFY_IS_TRUE(HasDxrInvocationLogEntryCountCheck(Lines, 1)); - VERIFY_IS_FALSE(HasDxrInvocationLogEntryCountCheck(Lines, 10)); + VERIFY_IS_TRUE(HasDxrInvocationLogEntryCountCheck(lines, 1)); } TEST_F(PixTest, DxilPIXDXRInvocationsLog_ExactCapacityUsesEntryCountBound) { - CComPtr CompiledLib = + auto compiledLib = Compile(m_dllSupport, kSingleMissInvocationLogShader, L"lib_6_6", {}); - CComPtr Output = RunDxilPIXDXRInvocationsLog(CompiledLib, 24); - const std::vector Lines = Tokenize(Disassemble(Output), "\n"); + auto output = RunDxilPIXDXRInvocationsLog(compiledLib, 24); + auto lines = Tokenize(Disassemble(output), "\n"); - VERIFY_IS_TRUE(HasDxrInvocationLogEntryCountCheck(Lines, 24)); - VERIFY_IS_FALSE(HasDxrInvocationLogEntryCountCheck(Lines, 240)); + VERIFY_IS_TRUE(HasDxrInvocationLogEntryCountCheck(lines, 24)); } TEST_F(PixTest, DxilPIXDXRInvocationsLog_OverflowGuardValidates) { - CComPtr CompiledLib = + auto compiledLib = Compile(m_dllSupport, kSingleMissInvocationLogShader, L"lib_6_6", {}); - CComPtr Output = RunDxilPIXDXRInvocationsLog(CompiledLib, 1); - const std::string Disassembly = Disassemble(Output); + auto output = RunDxilPIXDXRInvocationsLog(compiledLib, 1); + std::string disassembly = Disassemble(output); - VERIFY_IS_TRUE(Disassembly.find("@dx.op.binary.i32") == std::string::npos); - verifyInstrumentedModuleIsValid(Output, "DXR invocations log overflow guard"); + VERIFY_IS_TRUE(disassembly.find("@dx.op.binary.i32") == std::string::npos); + verifyInstrumentedModuleIsValid(output, "DXR invocations log overflow guard"); } uint32_t NuriGetWaveInstructionCount(const std::vector &lines) { @@ -6096,3 +6298,311 @@ float4 main(float4 pos : SV_Position) : SV_Target auto output = RunPixelHitPass(compiled, 16, 64, 0 /*requiredSVPositionRow*/); verifyInstrumentedModuleIsValid(output.blob, "pixel-hit instrumentation"); } +static const char *const kHullShaderWithHelper = R"x( +struct HsConstantData +{ + float Edges[3] : SV_TessFactor; + float Inside : SV_InsideTessFactor; +}; + +struct ControlPoint +{ + float3 position : WORLDPOS; +}; + +struct OutputPoint +{ + float3 vPosition : BEZIERPOS; +}; + +[noinline] +float HullHelper(float value) +{ + return value * 2.f; +} + +HsConstantData PatchConstantFunction(InputPatch ip) +{ + HsConstantData Output; + Output.Edges[0] = HullHelper(ip[0].position.x); + Output.Edges[1] = 8; + Output.Edges[2] = 8; + Output.Inside = 8; + return Output; +} + +[domain("tri")] +[partitioning("integer")] +[outputtopology("triangle_cw")] +[outputcontrolpoints(3)] +[patchconstantfunc("PatchConstantFunction")] +OutputPoint main(InputPatch ip, uint i : SV_OutputControlPointID) +{ + OutputPoint Output; + Output.vPosition = ip[i].position * HullHelper(2.f); + return Output; +})x"; + +static const char *const kComputeShaderWithHelper = R"x( +RWStructuredBuffer Output : register(u0); + +[noinline] +float ScaleHelper(float value) +{ + float scaled = value * 3.f; + return scaled; +} + +[numthreads(1, 1, 1)] +void main(uint3 threadId : SV_DispatchThreadID) +{ + Output[threadId.x] = ScaleHelper(threadId.y); +})x"; + +// A module names no entry function, so nothing roots the call graph. The +// inlining leaves such a module alone rather than erase every function in it. +// The entry point has no caller in the IR, so it is the first body that an +// inlining rooted on the patch-constant function alone erases. +TEST_F(PixTest, HelperInlining_NoEntryFunctionLeavesModuleAlone) { + auto compiled = + Compile(m_dllSupport, kHullShaderWithHelper, L"hs_6_2", {L"-Od"}); + std::string disassembly = Disassemble(compiled); + + // The baseline names an entry function, so the edit below is what removes it. + VERIFY_IS_TRUE(disassembly.find("!{void ()* @main,") != std::string::npos); + std::string withoutEntry = + ReplaceOnlyOccurrence(disassembly, "!{void ()* @main,", "!{null,"); + + std::vector lines = RunAnnotationPassOnText(withoutEntry); + + // Every function is still here, entry point included. + VERIFY_IS_TRUE(AnyLineContains(lines, "define void @main()")); + VERIFY_IS_TRUE(AnyLineContains(lines, "HullHelper")); + VERIFY_IS_TRUE(AnyLineContains(lines, "PatchConstantFunction")); +} + +// Something other than a direct call reaches a function, so the function +// survives inlining. The annotation pass then advertises it to PIX as a +// steppable range that no trace record arrives for, and names it in the report +// so PIX does not offer it. Taking the address of a helper produces that shape, +// which HLSL itself does not express. +TEST_F(PixTest, HelperInlining_SurvivingHelperIsReportedByExactName) { + auto compiled = + Compile(m_dllSupport, kComputeShaderWithHelper, L"cs_6_2", {L"-Od"}); + std::string withAddressTaken = + TakeAddressOfHelper(Disassemble(compiled), "ScaleHelper@@YAMM@Z"); + + std::vector lines = RunAnnotationPassOnText(withAddressTaken); + + // Exactly one record arrives, and it names the helper exactly. The report + // strips the leading mangling marker that + // PrintableSubsetOfMangledFunctionName removes, so the whole payload is + // compared, not a substring of it. + std::vector records = + RecordsWithPrefix(lines, "UninlinedFunction:"); + VERIFY_ARE_EQUAL(1u, static_cast(records.size())); + VERIFY_ARE_EQUAL(std::string("ScaleHelper@@YAMM@Z"), records[0]); + + // The record is deterministic, so a second run over the same input produces + // the same one record. + std::vector repeatRecords = RecordsWithPrefix( + RunAnnotationPassOnText(withAddressTaken), "UninlinedFunction:"); + VERIFY_ARE_EQUAL(records, repeatRecords); +} + +// Both prepasses inline, and the debug pipeline runs the dbg-value pass first, +// so the annotation pass reaches an already-inlined module and has nothing left +// to inline. It still names the survivor, because it is the pass that +// advertises the instruction range the survivor keeps. +TEST_F(PixTest, HelperInlining_SurvivingHelperIsReportedAfterEarlierPrepass) { + auto compiled = + Compile(m_dllSupport, kComputeShaderWithHelper, L"cs_6_2", {L"-Od"}); + std::string withAddressTaken = + TakeAddressOfHelper(Disassemble(compiled), "ScaleHelper@@YAMM@Z"); + + std::vector annotateOnly = RecordsWithPrefix( + RunAnnotationPassOnText(withAddressTaken), "UninlinedFunction:"); + VERIFY_ARE_EQUAL(1u, static_cast(annotateOnly.size())); + + // The report survives the earlier prepass, and arrives once rather than once + // per pass that inlines. + std::vector bothPrepasses = + RecordsWithPrefix(RunDbgValueAndAnnotationPassesOnText(withAddressTaken), + "UninlinedFunction:"); + VERIFY_ARE_EQUAL(1u, static_cast(bothPrepasses.size())); + VERIFY_ARE_EQUAL(annotateOnly, bothPrepasses); + + // The range the report is about is advertised, so the report is not vacuous. + std::vector ranges = + RecordsWithPrefix(RunDbgValueAndAnnotationPassesOnText(withAddressTaken), + "InstructionRange: "); + VERIFY_ARE_EQUAL(2u, static_cast(ranges.size())); + VERIFY_IS_TRUE(AnyLineContains(ranges, "ScaleHelper")); +} + +// An inlined helper produces no record at all, so the presence of a record +// means what the test above asserts it means. +TEST_F(PixTest, HelperInlining_InlinedHelperIsNotReported) { + auto compiled = + Compile(m_dllSupport, kComputeShaderWithHelper, L"cs_6_2", {L"-Od"}); + std::vector lines = + RunAnnotationPassOnText(Disassemble(compiled)); + + VERIFY_ARE_EQUAL(0u, + static_cast( + RecordsWithPrefix(lines, "UninlinedFunction:").size())); + + // The report stream also carries the module, so the same lines show that the + // helper has no body left to advertise. + for (const std::string &line : lines) { + if (line.compare(0, strlen("define "), "define ") == 0) { + VERIFY_IS_TRUE(line.find("ScaleHelper") == std::string::npos); + } + } +} + +// The runtime invokes a hull shader patch-constant function rather than the +// entry point does, so it survives inlining and needs instrumentation of its +// own. The helper that both of them call is inlined away, and the emitted DXIL +// validates. +TEST_F(PixTest, HelperInlining_HullPatchConstantFunctionValidates) { + auto compiled = + Compile(m_dllSupport, kHullShaderWithHelper, L"hs_6_2", {L"-Od"}); + auto output = RunDebugPass(compiled); + std::string disassembly = Disassemble(output.blob); + + // Two ranges are advertised, one of them the patch-constant function, so the + // helper is inlined away and nothing else is offered. + std::vector ranges = + RecordsWithPrefix(output.lines, "InstructionRange: "); + VERIFY_ARE_EQUAL(2u, static_cast(ranges.size())); + VERIFY_IS_TRUE(AnyLineContains(ranges, "PatchConstantFunction")); + VERIFY_IS_TRUE(AnyLineContains(ranges, "main hs")); + VERIFY_ARE_EQUAL( + 0u, static_cast( + RecordsWithPrefix(output.lines, "UninlinedFunction:").size())); + + VERIFY_ARE_EQUAL(0u, CountFunctionDefinitions(disassembly, "HullHelper")); + + // Both functions carry a selection prolog, so PIX receives trace records for + // either of them. + VERIFY_ARE_EQUAL(2u, CountOccurrences(disassembly, "PIXInterestingBlock:")); + + // OutputControlPointID is legal only in the control point phase, so the + // patch-constant function selects on the primitive alone. + std::string patchConstantBody = + ExtractFunctionBody(disassembly, "PatchConstantFunction"); + VERIFY_ARE_EQUAL( + 0u, CountCallsTo(patchConstantBody, "@dx.op.outputControlPointID.i32")); + VERIFY_ARE_EQUAL(1u, + CountCallsTo(patchConstantBody, "@dx.op.primitiveID.i32")); + + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of a hull shader whose " + "patch-constant function is instrumented too"); +} + +// The non-uniform-resource-index pipeline shares the annotation prepass, so it +// also sees inlined shaders. A dynamic index inside a helper is still +// diagnosed, and the emitted DXIL validates. +TEST_F(PixTest, HelperInlining_NonUniformResourceIndexHelperValidates) { + const char *source = R"x( +Texture2D tex[8] : register(t0); + +[noinline] +float IndexInHelper(float u, float v) +{ + uint index = u * v; + return tex[index].Load(int3(0, 0, 0)).x; +} + +float4 main(float2 uv : TEXCOORD0) : SV_TARGET +{ + return IndexInHelper(uv.x, uv.y); +})x"; + + auto compiled = Compile(m_dllSupport, source, L"ps_6_0", {L"-Od"}); + std::string outputText; + auto output = + RunDxilNonUniformResourceIndexInstrumentation(compiled, outputText); + std::string disassembly = Disassemble(output.blob); + + VERIFY_ARE_EQUAL(0u, CountFunctionDefinitions(disassembly, "IndexInHelper")); + + // The index is still reported, and the instrumentation is emitted. + VERIFY_IS_TRUE(outputText.find("FoundDynamicIndexingNoNuri") != + std::string::npos); + VERIFY_IS_TRUE( + outputText.find("NuriNotInstrumentedMissingInstructionNumber") == + std::string::npos); + VERIFY_ARE_EQUAL(1u, CountCallsTo(disassembly, "@dx.op.waveActiveAllEqual")); + + verifyInstrumentedModuleIsValid( + output.blob, "non-uniform resource index instrumentation of a dynamic " + "index inside an inlined helper"); +} + +// The debug-break pipeline shares the annotation prepass too. A DebugBreak +// inside a helper is still found once the helper is part of the entry point, +// and the emitted DXIL validates. +TEST_F(PixTest, HelperInlining_DebugBreakHelperValidates) { + const char *source = R"x( +RWStructuredBuffer Output : register(u0); + +[noinline] +uint BreakInHelper(uint value) +{ + DebugBreak(); + return value + 1; +} + +[numthreads(1, 1, 1)] +void main(uint3 threadId : SV_DispatchThreadID) +{ + Output[0] = BreakInHelper(threadId.x); +})x"; + + if (m_ver.SkipDxilVersion(1, 10)) { + return; + } + + auto compiled = Compile(m_dllSupport, source, L"cs_6_10", {L"-Od"}); + auto output = RunDebugBreakPass(compiled); + std::string disassembly = Disassemble(output.blob); + + VERIFY_ARE_EQUAL(0u, CountFunctionDefinitions(disassembly, "BreakInHelper")); + + // The break is recorded from inside the entry point, and the original call is + // gone. + VERIFY_ARE_EQUAL(1u, CountCallsTo(disassembly, "@dx.op.atomicBinOp.i32")); + VERIFY_ARE_EQUAL(0u, CountCallsTo(disassembly, "@dx.op.debugBreak")); + + verifyInstrumentedModuleIsValid( + output.blob, "debug-break instrumentation of a break inside an inlined " + "helper"); +} + +// The whole debug pipeline runs over a shader with a [noinline] helper. The +// helper becomes part of the entry point, one instruction range is advertised, +// the helper local is still traced, and the emitted DXIL validates. +TEST_F(PixTest, HelperInlining_InlinedHelperValidates) { + auto compiled = + Compile(m_dllSupport, kComputeShaderWithHelper, L"cs_6_2", {L"-Od"}); + auto output = RunDebugPass(compiled); + std::string disassembly = Disassemble(output.blob); + + // One function gives PIX one invocation identity. + std::vector ranges = + RecordsWithPrefix(output.lines, "InstructionRange: "); + VERIFY_ARE_EQUAL(1u, static_cast(ranges.size())); + VERIFY_IS_TRUE(ranges[0].find("main cs") != std::string::npos); + VERIFY_ARE_EQUAL( + 0u, static_cast( + RecordsWithPrefix(output.lines, "UninlinedFunction:").size())); + + VERIFY_ARE_EQUAL(0u, CountFunctionDefinitions(disassembly, "ScaleHelper")); + + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of a shader whose [noinline] helper " + "is inlined away"); +} diff --git a/tools/clang/unittests/HLSLExec/ExecutionTest.cpp b/tools/clang/unittests/HLSLExec/ExecutionTest.cpp index 5027f8daac..23a98a2f5e 100644 --- a/tools/clang/unittests/HLSLExec/ExecutionTest.cpp +++ b/tools/clang/unittests/HLSLExec/ExecutionTest.cpp @@ -10947,6 +10947,13 @@ void ExecutionTest::GroupWaveIndexTest() { WEX::TestExecution::SetVerifyOutput VerifySettings( WEX::TestExecution::VerifyOutputSettings::LogOnlyFailures); + BEGIN_TEST_METHOD_PROPERTIES() + TEST_METHOD_PROPERTY(L"Kits.TestId", L"c3f60f00-8e91-4acb-b4be-9f483fbe836b") + TEST_METHOD_PROPERTY( + L"Kits.Specification", + L"Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") + END_TEST_METHOD_PROPERTIES() + bool FailIfRequirementsNotMet = false; #ifdef _HLK_CONF FailIfRequirementsNotMet = true; diff --git a/tools/clang/unittests/HLSLExec/LinAlgTests.cpp b/tools/clang/unittests/HLSLExec/LinAlgTests.cpp index 0e19508fc9..4f4bcc8de7 100644 --- a/tools/clang/unittests/HLSLExec/LinAlgTests.cpp +++ b/tools/clang/unittests/HLSLExec/LinAlgTests.cpp @@ -25,7 +25,6 @@ #include "HlslTestUtils.h" #include -#include #include #include #include @@ -2935,62 +2934,6 @@ static void reportMissingConversionApi(LPCWSTR CaseName) { #endif // !defined(HLSLEXEC_LINALG_HOST_API) } // namespace matvec_interpretation - -namespace cpu_oracle { - -struct FloatToIntCase { - float Input; - int16_t Lower; - int16_t Upper; -}; - -static const std::vector F32ToI16Cases = { - {-40000.0f, -32768, -32768}, - {-32768.5f, -32768, -32768}, - {-2.5f, -3, -2}, - {-1.5f, -2, -1}, - {1.5f, 1, 2}, - {2.5f, 2, 3}, - {32767.5f, 32767, 32767}, - {40000.0f, 32767, 32767}, - {-2.75f, -3, -3}, - {-2.25f, -2, -2}, - {-1.75f, -2, -2}, - {-1.25f, -1, -1}, - {1.25f, 1, 1}, - {1.75f, 2, 2}, - {2.25f, 2, 2}, - {2.75f, 3, 3}, - {-255.5f, -256, -255}, - {255.5f, 255, 256}, - {-32767.5f, -32768, -32767}, - {32766.5f, 32766, 32767}, - {-0.5f, -1, 0}, - {0.5f, 0, 1}, - {-0.0f, 0, 0}, - {0.0f, 0, 0}, - {-std::numeric_limits::infinity(), -32768, -32768}, - {std::numeric_limits::infinity(), 32767, 32767}, - {-std::numeric_limits::quiet_NaN(), 0, 0}, - {std::numeric_limits::quiet_NaN(), 0, 0}, - {-3.25f, -3, -3}, - {-3.75f, -4, -4}, - {3.25f, 3, 3}, - {3.75f, 4, 4}, -}; - -static bool isNearestSaturatedI16(float Input, int16_t Actual) { - if (std::isnan(Input)) - return Actual == 0; - if (Input <= std::numeric_limits::min()) - return Actual == std::numeric_limits::min(); - if (Input >= std::numeric_limits::max()) - return Actual == std::numeric_limits::max(); - return std::abs(Input - static_cast(Actual)) <= 0.5f; -} - -} // namespace cpu_oracle - // Harness self-check for the CPU oracle. Deliberately carries no Kits metadata // so HLK runs never select it; drivers are not certified against this class. class LinAlgCPUOracleTests { @@ -3010,28 +2953,8 @@ class LinAlgCPUOracleTests { TEST_METHOD(MatchedFP8MatVecHostOracle); TEST_METHOD(FP8HostOracle); TEST_METHOD(FP8MatrixValueEncoding); - TEST_METHOD(FloatToIntHostOracle); }; -void LinAlgCPUOracleTests::FloatToIntHostOracle() { - for (const cpu_oracle::FloatToIntCase &Case : cpu_oracle::F32ToI16Cases) { - bool Success = true; - for (int Candidate = std::numeric_limits::min(); - Candidate <= std::numeric_limits::max(); ++Candidate) { - const bool Expected = Candidate >= Case.Lower && Candidate <= Case.Upper; - if (cpu_oracle::isNearestSaturatedI16( - Case.Input, static_cast(Candidate)) != Expected) { - hlsl_test::LogErrorFmt( - L"Float-to-I16 oracle: input=%g, candidate=%d, permitted=[%d,%d]", - static_cast(Case.Input), Candidate, Case.Lower, Case.Upper); - Success = false; - break; - } - } - VERIFY_IS_TRUE(Success); - } -} - void LinAlgCPUOracleTests::ComponentByteSize() { struct ComponentSizes { ComponentType Type; @@ -3721,41 +3644,25 @@ void LinAlgCapabilityTests::CapabilityPolicyAndPredicates() { linalg_abi::D3D12_LINEAR_ALGEBRA_DATATYPE_FLOAT16); } -class LinAlgTestClassCommon { -public: - bool setupClass(); - bool setupMethod(); - -protected: - CComPtr D3DDevice; - dxc::SpecificDllLoader DxcSupport; - bool VerboseLogging = false; - bool Initialized = false; - std::optional D3D12SDK; - - WEX::TestExecution::SetVerifyOutput VerifyOutput{ - WEX::TestExecution::VerifyOutputSettings::LogOnlyFailures}; -}; - -class DxilConf_SM610_LinAlg_DescriptorIO : public LinAlgTestClassCommon { +class DxilConf_SM610_LinAlg { public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_DescriptorIO) + BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg) TEST_CLASS_PROPERTY("Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Descriptor I/O") - TEST_CLASS_PROPERTY("Kits.TestId", "e2f563d7-7fea-42c1-a841-3fb0decb43a7") - TEST_CLASS_PROPERTY("Kits.Description", - "Validates SM 6.10 linear algebra descriptor operations") + "D3D12 - Shader Model 6.10 - LinAlg Matrix Operations") + TEST_CLASS_PROPERTY("Kits.TestId", "a1b2c3d4-e5f6-7890-abcd-ef1234567890") + TEST_CLASS_PROPERTY( + "Kits.Description", + "Validates SM 6.10 linear algebra matrix operations execute correctly") TEST_CLASS_PROPERTY( "Kits.Specification", "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") TEST_METHOD_PROPERTY(L"Priority", L"0") END_TEST_CLASS() - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } + TEST_CLASS_SETUP(setupClass); + TEST_METHOD_SETUP(setupMethod); + // Load/Store/Accumulate Descriptor TEST_METHOD(LoadStoreDescriptor_Wave_16x16_F16); TEST_METHOD(LoadStoreDescriptor_Wave_4x8_F16_RowMajorOffsetPadded); TEST_METHOD(LoadStoreDescriptor_Wave_16x16_F16_RowMajorOffsetPadded); @@ -3775,28 +3682,8 @@ class DxilConf_SM610_LinAlg_DescriptorIO : public LinAlgTestClassCommon { TEST_METHOD(AccumulateDescriptor_Wave_16x16_F16); TEST_METHOD(AccumulateDescriptorContention_Wave_4x8_I32); TEST_METHOD(AccumulateDescriptorContention_Wave_4x8_F32_OrderInvariant); -}; - -class DxilConf_SM610_LinAlg_GroupSharedIO : public LinAlgTestClassCommon { -public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_GroupSharedIO) - TEST_CLASS_PROPERTY("Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Group-Shared I/O") - TEST_CLASS_PROPERTY("Kits.TestId", "5b3cdf07-cfe4-43a4-be25-9ea41b854334") - TEST_CLASS_PROPERTY( - "Kits.Description", - "Validates SM 6.10 linear algebra group-shared memory operations") - TEST_CLASS_PROPERTY( - "Kits.Specification", - "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") - TEST_METHOD_PROPERTY(L"Priority", L"0") - END_TEST_CLASS() - - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } + // Load/Store/Accumulate Memory TEST_METHOD(LoadMemory_Wave_16x16_F16); TEST_METHOD(StoreMemory_Wave_16x16_F16); TEST_METHOD(AccumulateMemory_Wave_16x16_F16); @@ -3810,97 +3697,22 @@ class DxilConf_SM610_LinAlg_GroupSharedIO : public LinAlgTestClassCommon { TEST_METHOD(AccumulateMemoryContention_Wave_4x8_F16); TEST_METHOD(AccumulateMemoryContention_Wave_16x16_F16); TEST_METHOD(AccumulateMemoryContention_Wave_4x8_I32); -}; - -class DxilConf_SM610_LinAlg_ElementAccess : public LinAlgTestClassCommon { -public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_ElementAccess) - TEST_CLASS_PROPERTY("Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Element Access") - TEST_CLASS_PROPERTY("Kits.TestId", "547bca7a-05e3-4c88-bbc9-8af88567fae3") - TEST_CLASS_PROPERTY( - "Kits.Description", - "Validates SM 6.10 linear algebra element access and matrix values") - TEST_CLASS_PROPERTY( - "Kits.Specification", - "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") - TEST_METHOD_PROPERTY(L"Priority", L"0") - END_TEST_CLASS() - - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } + // Element access TEST_METHOD(ElementAccess_Wave_16x16_F16); TEST_METHOD(ElementAccess_Wave_4x8_F32); TEST_METHOD(ElementSet_Wave_16x16_F16); - TEST_METHOD(MatrixCopy_Wave_16x16_F16); TEST_METHOD(ElementGetOOB_Wave_4x8_F32); TEST_METHOD(ElementSetOOB_Wave_4x8_F32); TEST_METHOD(ElementGetOOB_Wave_16x16_F16); TEST_METHOD(ElementSetOOB_Wave_16x16_F16); -}; - -class DxilConf_SM610_LinAlg_Conversion : public LinAlgTestClassCommon { -public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_Conversion) - TEST_CLASS_PROPERTY("Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Conversion") - TEST_CLASS_PROPERTY("Kits.TestId", "77de35e0-610e-4f23-aee4-b2d7302a0d98") - TEST_CLASS_PROPERTY( - "Kits.Description", - "Validates SM 6.10 linear algebra copy and conversion operations") - TEST_CLASS_PROPERTY( - "Kits.Specification", - "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") - TEST_METHOD_PROPERTY(L"Priority", L"0") - END_TEST_CLASS() - - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } + // Cast/Convert TEST_METHOD(CopyConvert_Wave_16x16_F16); TEST_METHOD(CopyConvert_Wave_16x16_F16_Transpose); TEST_METHOD(CopyConvert_Wave_4x8_F32_Transpose); - TEST_METHOD(Convert); - TEST_METHOD(CopyConvert_Wave_4x8_F16_ToF32); - TEST_METHOD(CopyConvert_Wave_4x8_F32_ToF16_Transpose); - TEST_METHOD(Convert_I16_ToI32_Exact); - TEST_METHOD(Convert_I32_ToI16_Saturate); - TEST_METHOD(Convert_I32_ToU16_Saturate); - TEST_METHOD(Convert_U32_ToI16_Saturate); - TEST_METHOD(Convert_U32_ToU16_Saturate); - TEST_METHOD(Convert_F32_ToI16_RoundNearest_Saturate); - TEST_METHOD(Convert_F32_ToI16_NonFinite); - TEST_METHOD(Convert_I32_ToF16_RTNE); - TEST_METHOD(Convert_F16_ToE4M3FN_AndBack); - TEST_METHOD(Convert_F16_ToE5M2_AndBack); -}; - -class DxilConf_SM610_LinAlg_MatrixArithmetic : public LinAlgTestClassCommon { -public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_MatrixArithmetic) - TEST_CLASS_PROPERTY("Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Matrix Arithmetic") - TEST_CLASS_PROPERTY("Kits.TestId", "fac50b41-fa60-4d37-8172-a8e90f88874b") - TEST_CLASS_PROPERTY( - "Kits.Description", - "Validates SM 6.10 linear algebra matrix arithmetic operations") - TEST_CLASS_PROPERTY( - "Kits.Specification", - "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") - TEST_METHOD_PROPERTY(L"Priority", L"0") - END_TEST_CLASS() - - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } - + // Matrix Matrix Arithmetic TEST_METHOD(MatMatMul_Wave_16x16x16_F16); TEST_METHOD(MatMatMul_Wave_8x32x16_F16_NonUniform); TEST_METHOD(MatMatMul_Wave_8x32x16_F16_ToF32); @@ -3918,36 +3730,11 @@ class DxilConf_SM610_LinAlg_MatrixArithmetic : public LinAlgTestClassCommon { TEST_METHOD(MatAccum_Wave_8x32_F16_BUse_NonUniform); TEST_METHOD(MatAccum_Wave_16x32_F16_BUse_NonUniform); - TEST_METHOD(QueryAccumLayout); -}; - -class DxilConf_SM610_LinAlg_MatVec : public LinAlgTestClassCommon { -public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_MatVec) - TEST_CLASS_PROPERTY("Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Matrix-Vector") - TEST_CLASS_PROPERTY("Kits.TestId", "6c8121c0-42e3-4e10-a844-5f601a3f1a33") - TEST_CLASS_PROPERTY( - "Kits.Description", - "Validates SM 6.10 linear algebra matrix-vector operations") - TEST_CLASS_PROPERTY( - "Kits.Specification", - "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") - TEST_METHOD_PROPERTY(L"Priority", L"0") - END_TEST_CLASS() - - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } - + // Matrix Vector Arithmetic TEST_METHOD(MatVecMul_Thread_16x16_F16); TEST_METHOD(MatVecMul_Thread_4x8_F32); TEST_METHOD(MatVecMul_Thread_4x8_F16_PerThread); TEST_METHOD(MatVecMul_Thread_4x8_F16_Divergent); - TEST_METHOD(MatVecMul_Thread_4x8_F16_MatrixArray); - TEST_METHOD(MatVecMul_Thread_4x8_F16_MatrixPhi); - TEST_METHOD(MatVecMul_Thread_4x8_F16_MatrixSelect); TEST_METHOD(MatVecMul_Thread_4x8_F16_NonUniform); TEST_METHOD(MatVecMul_Thread_4x8_F16_ColumnMajor); TEST_METHOD(MatVecMul_Thread_4x8_I8_Interpreted); @@ -3966,31 +3753,28 @@ class DxilConf_SM610_LinAlg_MatVec : public LinAlgTestClassCommon { TEST_METHOD(MatVecMulAdd_Thread_16x16_F16); TEST_METHOD(MatVecMulAdd_Thread_4x8_F32); TEST_METHOD(MatVecMulAdd_Thread_4x8_F16_IndependentBias); -}; + TEST_METHOD(OuterProduct_Thread_16x16_F16); -class DxilConf_SM610_LinAlg_OuterVectorAccumulation - : public LinAlgTestClassCommon { -public: - BEGIN_TEST_CLASS(DxilConf_SM610_LinAlg_OuterVectorAccumulation) - TEST_CLASS_PROPERTY( - "Kits.TestName", - "D3D12 - Shader Model 6.10 - LinAlg Outer and Vector Accumulation") - TEST_CLASS_PROPERTY("Kits.TestId", "c4d69b38-6eaa-4dde-bd46-e1e578ceae09") - TEST_CLASS_PROPERTY( - "Kits.Description", - "Validates SM 6.10 linear algebra outer-product and vector accumulation") - TEST_CLASS_PROPERTY( - "Kits.Specification", - "Device.Graphics.D3D12.DXILCore.ShaderModel610.CoreRequirement") - TEST_METHOD_PROPERTY(L"Priority", L"0") - END_TEST_CLASS() + // Query Accumulator Layout + TEST_METHOD(QueryAccumLayout); - TEST_CLASS_SETUP(setupClass) { return LinAlgTestClassCommon::setupClass(); } - TEST_METHOD_SETUP(setupMethod) { - return LinAlgTestClassCommon::setupMethod(); - } + // Convert + TEST_METHOD(Convert); - TEST_METHOD(OuterProduct_Thread_16x16_F16); + // CopyConvert / Convert coverage + TEST_METHOD(CopyConvert_Wave_4x8_F16_ToF32); + TEST_METHOD(CopyConvert_Wave_4x8_F32_ToF16_Transpose); + TEST_METHOD(Convert_I16_ToI32_Exact); + TEST_METHOD(Convert_I32_ToI16_Saturate); + TEST_METHOD(Convert_I32_ToU16_Saturate); + TEST_METHOD(Convert_U32_ToI16_Saturate); + TEST_METHOD(Convert_U32_ToU16_Saturate); + TEST_METHOD(Convert_F32_ToI16_RTNE_Saturate); + TEST_METHOD(Convert_I32_ToF16_RTNE); + TEST_METHOD(Convert_F16_ToE4M3FN_AndBack); + TEST_METHOD(Convert_F16_ToE5M2_AndBack); + + // Vector Accumulate TEST_METHOD(VectorAccumulateDescriptor_Thread_F16); TEST_METHOD(VectorAccumulateDescriptor_Thread_F16_Length8_NonZero); TEST_METHOD(VectorAccumulateDescriptor_Thread_F32_Length8_NonZero); @@ -3999,9 +3783,19 @@ class DxilConf_SM610_LinAlg_OuterVectorAccumulation TEST_METHOD(VectorAccumulateDescriptorContention_Thread_F16); TEST_METHOD(VectorAccumulateDescriptorContention_Thread_F32_OrderInvariant); TEST_METHOD(VectorAccumulateDescriptorContention_Thread_I32); + +private: + CComPtr D3DDevice; + dxc::SpecificDllLoader DxcSupport; + bool VerboseLogging = false; + bool Initialized = false; + std::optional D3D12SDK; + + WEX::TestExecution::SetVerifyOutput VerifyOutput{ + WEX::TestExecution::VerifyOutputSettings::LogOnlyFailures}; }; -bool LinAlgTestClassCommon::setupClass() { +bool DxilConf_SM610_LinAlg::setupClass() { if (!Initialized) { Initialized = true; VERIFY_SUCCEEDED( @@ -4026,7 +3820,7 @@ bool LinAlgTestClassCommon::setupClass() { return true; } -bool LinAlgTestClassCommon::setupMethod() { +bool DxilConf_SM610_LinAlg::setupMethod() { // If the device is healthy, exit otherwise it's possible a previous test // case caused a device removal. So we need to try and create a new device. if (D3DDevice && D3DDevice->GetDeviceRemovedReason() == S_OK) @@ -4511,7 +4305,7 @@ static constexpr size_t DescriptorAlignedOffset = MatrixOffsetAlignmentBytes; static_assert(DescriptorAlignedOffset % DescriptorDeclaredAlignment == 0, "descriptor offset must keep the first element aligned"); -void DxilConf_SM610_LinAlg_DescriptorIO::LoadStoreDescriptor_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::LoadStoreDescriptor_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -4538,7 +4332,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO::LoadStoreDescriptor_Wave_16x16_F16() { // three 16-byte gaps between its four rows. A store that addresses by element // index rather than by the supplied stride writes into that padding, which the // untouched-byte check catches and the element comparison cannot. -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadStoreDescriptor_Wave_4x8_F16_RowMajorOffsetPadded() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4569,7 +4363,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: VerboseLogging, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadStoreDescriptor_Wave_16x16_F16_RowMajorOffsetPadded() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4604,7 +4398,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: // layout argument entirely still round trips byte-identically, because the // mapping it applies to the load it applies again to the store. Reading one // layout and writing the other stops the two from cancelling. -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadStoreDescriptor_Wave_4x8_F32_RowMajorToColumnMajor() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; @@ -4646,7 +4440,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: // must stay non-square: swapping the two layouts transposes on load and back // on store, and for a square matrix those cancel byte for byte whatever // strides are used. -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadStoreDescriptor_Wave_4x8_F16_RowMajorToColumnMajor() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4685,7 +4479,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: } // A rectangular multiple of a 16x16 tile prevents swapped layouts cancelling. -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadStoreDescriptor_Wave_16x32_F16_RowMajorToColumnMajor() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4724,8 +4518,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: // Half the source matrix lies outside the view the descriptor carries. The // boundary is deliberately placed mid-row rather than on a row boundary, so an // implementation that bounds checks a row at a time cannot pass it. -void DxilConf_SM610_LinAlg_DescriptorIO:: - LoadDescriptorOOB_Wave_16x16_F16_PartialView() { +void DxilConf_SM610_LinAlg::LoadDescriptorOOB_Wave_16x16_F16_PartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -4754,7 +4547,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: // the view boundary falls in a different place for byte offsets than it does // for element indices. An implementation that bounds checks by element index // keeps elements this view does not reach. -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadDescriptorOOB_Wave_4x8_F16_OffsetPaddedPartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4787,7 +4580,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: LoadDescriptorOOB_Wave_16x16_F16_OffsetPaddedPartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4821,8 +4614,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: // The same two views on the destination instead of the source, so the rule // being exercised is bounds checking on the store rather than on the load. -void DxilConf_SM610_LinAlg_DescriptorIO:: - StoreDescriptorOOB_Wave_16x16_F16_PartialView() { +void DxilConf_SM610_LinAlg::StoreDescriptorOOB_Wave_16x16_F16_PartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -4849,7 +4641,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: VerboseLogging, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: StoreDescriptorOOB_Wave_4x8_F16_OffsetPaddedPartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4882,7 +4674,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: StoreDescriptorOOB_Wave_16x16_F16_OffsetPaddedPartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4913,7 +4705,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: AccumulateDescriptorOOB_Wave_16x16_F16_PartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4943,7 +4735,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: /*OutputViewBytes=*/260, VerboseLogging, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: AccumulateDescriptorOOB_Wave_4x8_F16_OffsetPaddedPartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -4979,7 +4771,7 @@ void DxilConf_SM610_LinAlg_DescriptorIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: AccumulateDescriptorOOB_Wave_16x16_F16_OffsetPaddedPartialView() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -5071,7 +4863,7 @@ static void runSplatStore(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_DescriptorIO::SplatStore_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::SplatStore_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -5190,7 +4982,7 @@ static void runAccumulateDescriptor(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_DescriptorIO::AccumulateDescriptor_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::AccumulateDescriptor_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -5246,14 +5038,13 @@ static void runAccumulateDescriptorContention( SelectedWaveSize, ActiveWaveCount, DispatchX); } -void DxilConf_SM610_LinAlg_DescriptorIO:: - AccumulateDescriptorContention_Wave_4x8_I32() { +void DxilConf_SM610_LinAlg::AccumulateDescriptorContention_Wave_4x8_I32() { runAccumulateDescriptorContention( D3DDevice, DxcSupport, ComponentType::I32, /*FillValue=*/7, L"AccumulateDescriptorContention_Wave_4x8_I32", VerboseLogging); } -void DxilConf_SM610_LinAlg_DescriptorIO:: +void DxilConf_SM610_LinAlg:: AccumulateDescriptorContention_Wave_4x8_F32_OrderInvariant() { runAccumulateDescriptorContention( D3DDevice, DxcSupport, ComponentType::F32, /*FillValue=*/1, @@ -5365,7 +5156,7 @@ static void runElementAccess(ID3D12Device *Device, TotalLength, NumElements, "Sum of all lengths must be gte num elements"); } -void DxilConf_SM610_LinAlg_ElementAccess::ElementAccess_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::ElementAccess_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -5386,7 +5177,7 @@ void DxilConf_SM610_LinAlg_ElementAccess::ElementAccess_Wave_16x16_F16() { runElementAccess(D3DDevice, DxcSupport, Params, VerboseLogging, WaveSize); } -void DxilConf_SM610_LinAlg_ElementAccess::ElementAccess_Wave_4x8_F32() { +void DxilConf_SM610_LinAlg::ElementAccess_Wave_4x8_F32() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -5424,16 +5215,14 @@ static const char ElementSetShader[] = R"( if (GetGroupWaveIndex() != 0) return; - using Matrix = __builtin_LinAlgMatrix - [[__LinAlgMatrix_Attributes(COMP_TYPE, M_DIM, N_DIM, USE, SCOPE)]]; - Matrix Original; + __builtin_LinAlgMatrix + [[__LinAlgMatrix_Attributes(COMP_TYPE, M_DIM, N_DIM, USE, SCOPE)]] + Mat; dx::__builtin_LinAlg_MatrixLoadFromDescriptor( - Original, Input, 0, STRIDE, LAYOUT, 128); - Matrix Mat = Original; + Mat, Input, 0, STRIDE, LAYOUT, 128); // Increment every element by 5 - const uint Count = WaveActiveMax(dx::__builtin_LinAlg_MatrixLength(Mat)); - for (uint I = 0; I < Count; ++I) { + for (uint I = 0; I < dx::__builtin_LinAlg_MatrixLength(Mat); ++I) { ELEM_TYPE Elem; dx::__builtin_LinAlg_MatrixGetElement(Elem, Mat, I); Elem = Elem + 5; @@ -5442,27 +5231,19 @@ static const char ElementSetShader[] = R"( dx::__builtin_LinAlg_MatrixStoreToDescriptor( Mat, Output, 0, STRIDE, LAYOUT, 128); -#ifdef VERIFY_COPY_ISOLATION - dx::__builtin_LinAlg_MatrixStoreToDescriptor( - Original, Output, ORIGINAL_OFFSET, STRIDE, LAYOUT, 128); -#endif } )"; static void runElementSet(ID3D12Device *Device, dxc::SpecificDllLoader &DxcSupport, const MatrixParams &Params, bool Verbose, - UINT ForcedWaveSize = 0, bool VerifyCopy = false) { + UINT ForcedWaveSize = 0) { const size_t NumElements = Params.totalElements(); const size_t MatrixSize = Params.totalBytes(); - const size_t OutputBytes = - VerifyCopy ? 2 * MatrixSize + MatrixStrideAlignmentBytes : MatrixSize; std::stringstream ExtraDefs; if (ForcedWaveSize != 0) ExtraDefs << " -DFORCED_WAVE_SIZE=" << ForcedWaveSize; - if (VerifyCopy) - ExtraDefs << " -DVERIFY_COPY_ISOLATION -DORIGINAL_OFFSET=" << MatrixSize; std::string Args = buildCompilerArgs(Params, ExtraDefs.str().c_str()); compileShader(DxcSupport, ElementSetShader, "cs_6_10", Args, Verbose); @@ -5473,19 +5254,14 @@ static void runElementSet(ID3D12Device *Device, auto Op = createComputeOp(ElementSetShader, "cs_6_10", "UAV(u0), UAV(u1)", Args.c_str()); addUAVBuffer(Op.get(), "Input", MatrixSize, false, "byname"); - addUAVBuffer(Op.get(), "Output", OutputBytes, true, - VerifyCopy ? "byname" : "zero"); + addUAVBuffer(Op.get(), "Output", MatrixSize, true); addRootView(Op.get(), 0, "Input"); addRootView(Op.get(), 1, "Output"); auto Result = runShaderOp(Device, DxcSupport, std::move(Op), - [NumElements, Params, VerifyCopy]( - LPCSTR Name, std::vector &Data, st::ShaderOp *) { - if (VerifyCopy && _stricmp(Name, "Output") == 0) { - cpu_oracle::fillPoison(Data.data(), Data.size()); - return; - } + [NumElements, Params](LPCSTR Name, std::vector &Data, + st::ShaderOp *) { VERIFY_IS_TRUE(fillInputBuffer(Name, Data, Params.CompType, NumElements), "Saw unsupported component type"); @@ -5493,27 +5269,13 @@ static void runElementSet(ID3D12Device *Device, MappedData OutData; Result->Test->GetReadBackData("Output", &OutData); - VERIFY_ARE_EQUAL(OutputBytes, OutData.size()); - if (OutputBytes != OutData.size()) - return; // Verify the front of the buffer is a list of elements of the expected type VERIFY_IS_TRUE(verifyComponentBuffer(Params.CompType, OutData.data(), Expected, NumElements, Verbose)); - if (VerifyCopy) { - const VariantCompType OriginalExpected = - makeExpectedMat(Params.CompType, Params.M, Params.N, 1); - VERIFY_IS_TRUE(verifyComponentBuffer( - Params.CompType, static_cast(OutData.data()) + MatrixSize, - OriginalExpected, NumElements, Verbose)); - VERIFY_IS_TRUE(cpu_oracle::verifyUntouchedBytes( - Params.CompType, Params.M * 2, Params.N, - {Params.Layout, 0, Params.strideBytes()}, OutData.data(), - OutData.size(), Verbose)); - } } -void DxilConf_SM610_LinAlg_ElementAccess::ElementSet_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::ElementSet_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -5534,28 +5296,6 @@ void DxilConf_SM610_LinAlg_ElementAccess::ElementSet_Wave_16x16_F16() { runElementSet(D3DDevice, DxcSupport, Params, VerboseLogging, WaveSize); } -void DxilConf_SM610_LinAlg_ElementAccess::MatrixCopy_Wave_16x16_F16() { - MatrixParams Params = {}; - Params.CompType = ComponentType::F16; - Params.M = 16; - Params.N = 16; - Params.Use = MatrixUse::Accumulator; - Params.Scope = MatrixScope::Wave; - Params.Layout = MatrixLayout::RowMajor; - Params.NumThreads = 128; - Params.Enable16Bit = true; - - std::vector WaveSizes; - if (!matrixConstructionApplicableWaveSizes(D3DDevice, Params, {Params.Use}, - L"MatrixCopy_Wave_16x16_F16", - WaveSizes)) - return; - - for (const UINT WaveSize : WaveSizes) - runElementSet(D3DDevice, DxcSupport, Params, VerboseLogging, WaveSize, - /*VerifyCopy=*/true); -} - // Length() is thread local, so the first index past a lane's own length is // already out of bounds even though the wave collectively holds more elements. // Probing Length() and a index far beyond it covers both a driver that clamps @@ -5722,7 +5462,7 @@ static void runElementGetOOB(ID3D12Device *Device, } } -void DxilConf_SM610_LinAlg_ElementAccess::ElementGetOOB_Wave_4x8_F32() { +void DxilConf_SM610_LinAlg::ElementGetOOB_Wave_4x8_F32() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -5839,7 +5579,7 @@ static void runElementSetOOB(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_ElementAccess::ElementSetOOB_Wave_4x8_F32() { +void DxilConf_SM610_LinAlg::ElementSetOOB_Wave_4x8_F32() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -5863,7 +5603,7 @@ void DxilConf_SM610_LinAlg_ElementAccess::ElementSetOOB_Wave_4x8_F32() { // Out-of-bounds element access on F16. Both cases above pin the boundary // behaviour to F32, which no tier is required to support, so a conforming // F16-only device would exercise neither. -void DxilConf_SM610_LinAlg_ElementAccess::ElementGetOOB_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::ElementGetOOB_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -5884,7 +5624,7 @@ void DxilConf_SM610_LinAlg_ElementAccess::ElementGetOOB_Wave_16x16_F16() { runElementGetOOB(D3DDevice, DxcSupport, Params, VerboseLogging, WaveSize); } -void DxilConf_SM610_LinAlg_ElementAccess::ElementSetOOB_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::ElementSetOOB_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -6142,7 +5882,7 @@ static void runCopyConvert(ID3D12Device *Device, SourceOracle, Verbose)); } -void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::CopyConvert_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -6164,7 +5904,7 @@ void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_16x16_F16() { /*Transpose=*/false, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_16x16_F16_Transpose() { +void DxilConf_SM610_LinAlg::CopyConvert_Wave_16x16_F16_Transpose() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -6187,7 +5927,7 @@ void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_16x16_F16_Transpose() { /*Transpose=*/true, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_4x8_F32_Transpose() { +void DxilConf_SM610_LinAlg::CopyConvert_Wave_4x8_F32_Transpose() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -6284,7 +6024,7 @@ static void runMatMatMul(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_MatrixArithmetic::MatMatMul_Wave_16x16x16_F16() { +void DxilConf_SM610_LinAlg::MatMatMul_Wave_16x16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -6381,8 +6121,7 @@ static void runMatMatMulAccum(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMulAccum_Wave_16x16x16_F16() { +void DxilConf_SM610_LinAlg::MatMatMulAccum_Wave_16x16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -6470,7 +6209,7 @@ static void runMatAccum(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_MatrixArithmetic::MatAccum_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::MatAccum_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -7390,24 +7129,21 @@ makeRectangularF16WaveMultiplyCase(MatrixDim M, ComponentType AccumulatorType, return Case; } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMul_Wave_8x32x16_F16_NonUniform() { +void DxilConf_SM610_LinAlg::MatMatMul_Wave_8x32x16_F16_NonUniform() { const MatrixMultiplyCase Case = makeRectangularF16WaveMultiplyCase( /*M=*/8, ComponentType::F16, MatrixMultiplyOperation::Multiply); runWaveMultiplyCase(D3DDevice, DxcSupport, Case, L"MatMatMul_Wave_8x32x16_F16_NonUniform", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMul_Wave_8x32x16_F16_ToF32() { +void DxilConf_SM610_LinAlg::MatMatMul_Wave_8x32x16_F16_ToF32() { const MatrixMultiplyCase Case = makeRectangularF16WaveMultiplyCase( /*M=*/8, ComponentType::F32, MatrixMultiplyOperation::Multiply); runWaveMultiplyCase(D3DDevice, DxcSupport, Case, L"MatMatMul_Wave_8x32x16_F16_ToF32", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMul_Wave_16x32x16_F16_NonUniform() { +void DxilConf_SM610_LinAlg::MatMatMul_Wave_16x32x16_F16_NonUniform() { const MatrixMultiplyCase Case = makeRectangularF16WaveMultiplyCase( /*M=*/16, ComponentType::F16, MatrixMultiplyOperation::Multiply); runWaveMultiplyCase(D3DDevice, DxcSupport, Case, @@ -7415,16 +7151,14 @@ void DxilConf_SM610_LinAlg_MatrixArithmetic:: VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMul_Wave_16x32x16_F16_ToF32() { +void DxilConf_SM610_LinAlg::MatMatMul_Wave_16x32x16_F16_ToF32() { const MatrixMultiplyCase Case = makeRectangularF16WaveMultiplyCase( /*M=*/16, ComponentType::F32, MatrixMultiplyOperation::Multiply); runWaveMultiplyCase(D3DDevice, DxcSupport, Case, L"MatMatMul_Wave_16x32x16_F16_ToF32", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMulAccum_Wave_8x32x16_F16_ToF32_NonUniform() { +void DxilConf_SM610_LinAlg::MatMatMulAccum_Wave_8x32x16_F16_ToF32_NonUniform() { const MatrixMultiplyCase Case = makeRectangularF16WaveMultiplyCase( /*M=*/8, ComponentType::F32, MatrixMultiplyOperation::MultiplyAccumulate); runWaveMultiplyCase(D3DDevice, DxcSupport, Case, @@ -7432,7 +7166,7 @@ void DxilConf_SM610_LinAlg_MatrixArithmetic:: VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: +void DxilConf_SM610_LinAlg:: MatMatMulAccum_Wave_16x32x16_F16_ToF32_NonUniform() { const MatrixMultiplyCase Case = makeRectangularF16WaveMultiplyCase( /*M=*/16, ComponentType::F32, @@ -7442,8 +7176,7 @@ void DxilConf_SM610_LinAlg_MatrixArithmetic:: VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMulAccum_Wave_16x16x16_F16_ToF32_BLayouts() { +void DxilConf_SM610_LinAlg::MatMatMulAccum_Wave_16x16x16_F16_ToF32_BLayouts() { MatrixMultiplyCase Case = {}; Case.MatrixAType = ComponentType::F16; Case.MatrixBType = ComponentType::F16; @@ -7468,7 +7201,7 @@ void DxilConf_SM610_LinAlg_MatrixArithmetic:: } } -void DxilConf_SM610_LinAlg_MatrixArithmetic::MatMatMul_Wave_16x16x16_I32() { +void DxilConf_SM610_LinAlg::MatMatMul_Wave_16x16x16_I32() { MatrixMultiplyCase Case = {}; Case.MatrixAType = ComponentType::I32; Case.MatrixBType = ComponentType::I32; @@ -7523,8 +7256,7 @@ makeRectangularF16ThreadGroupMultiplyPlan(ComponentType AccumulatorType, return Plan; } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMul_ThreadGroup_WaveScaled_F16_NonUniform() { +void DxilConf_SM610_LinAlg::MatMatMul_ThreadGroup_WaveScaled_F16_NonUniform() { const ThreadGroupMultiplyPlan Plan = makeRectangularF16ThreadGroupMultiplyPlan( ComponentType::F16, MatrixMultiplyOperation::Multiply); @@ -7533,7 +7265,7 @@ void DxilConf_SM610_LinAlg_MatrixArithmetic:: VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: +void DxilConf_SM610_LinAlg:: MatMatMulAccum_ThreadGroup_WaveScaled_F16_ToF32_NonUniform() { const ThreadGroupMultiplyPlan Plan = makeRectangularF16ThreadGroupMultiplyPlan( @@ -7544,8 +7276,7 @@ void DxilConf_SM610_LinAlg_MatrixArithmetic:: VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatMatMul_ThreadGroup_WaveScaled_I32() { +void DxilConf_SM610_LinAlg::MatMatMul_ThreadGroup_WaveScaled_I32() { ThreadGroupMultiplyPlan Plan = {}; Plan.MatrixAType = ComponentType::I32; Plan.MatrixBType = ComponentType::I32; @@ -7668,15 +7399,13 @@ static void runWaveAccumulateBUse(ID3D12Device *Device, L"Exact non-uniform F16 accumulator plus a B-use F16 matrix", Verbose)); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatAccum_Wave_8x32_F16_BUse_NonUniform() { +void DxilConf_SM610_LinAlg::MatAccum_Wave_8x32_F16_BUse_NonUniform() { runWaveAccumulateBUse(D3DDevice, DxcSupport, /*M=*/8, L"MatAccum_Wave_8x32_F16_BUse_NonUniform", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatrixArithmetic:: - MatAccum_Wave_16x32_F16_BUse_NonUniform() { +void DxilConf_SM610_LinAlg::MatAccum_Wave_16x32_F16_BUse_NonUniform() { runWaveAccumulateBUse(D3DDevice, DxcSupport, /*M=*/16, L"MatAccum_Wave_16x32_F16_BUse_NonUniform", VerboseLogging); @@ -7839,7 +7568,7 @@ static void runMatVecMul(ID3D12Device *Device, Expected, Params.M, Verbose)); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_16x16_F16() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -7862,7 +7591,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_16x16_F16() { /*FillValue=*/2, /*OutputSigned=*/true, ComponentType::F16); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F32() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F32() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -7975,9 +7704,10 @@ static std::vector buildThreadSemanticsVectorBuffer() { enum class ThreadSemanticsOutcome { Verified, Inconclusive, Failed }; -static ThreadSemanticsOutcome -verifyThreadSemanticsOutput(const void *Data, size_t Size, bool Divergent, - bool Verbose, bool MatrixSelection = false) { +static ThreadSemanticsOutcome verifyThreadSemanticsOutput(const void *Data, + size_t Size, + bool Divergent, + bool Verbose) { const size_t RequiredBytes = ThreadSemanticsThreads * ThreadSemanticsOutputSlotBytes; if (Size < RequiredBytes) { @@ -8018,7 +7748,7 @@ verifyThreadSemanticsOutput(const void *Data, size_t Size, bool Divergent, const bool UseVectorB = (Witness & ThreadSemanticsWitnessUseB) != 0; if (UseVectorB != (Arm == 1)) { hlsl_test::LogErrorFmt( - L"Thread %d executed arm %s but its predicate selected input %s", + L"Thread %d executed arm %s but its predicate selected vector %s", Thread, Arm == 1 ? L"B" : L"A", UseVectorB ? L"B" : L"A"); Success = false; continue; @@ -8033,23 +7763,19 @@ verifyThreadSemanticsOutput(const void *Data, size_t Size, bool Divergent, const HLSLHalf_t *Values = reinterpret_cast(Slot); for (int Row = 0; Row < ThreadSemanticsM; ++Row) { const float Actual = static_cast(Values[Row]); - const int MatrixThread = MatrixSelection && UseVectorB - ? ThreadSemanticsThreads - 1 - Thread - : Thread; - const float Expected = static_cast( - threadSemanticsExpected(MatrixThread, Row, UseVectorB)); + const float Expected = + static_cast(threadSemanticsExpected(Thread, Row, UseVectorB)); // Every expectation is an integer F16 holds exactly. if (Actual != Expected) { - hlsl_test::LogErrorFmt( - L"Thread %d row %d (matrix %d, vector %s): actual=%f, " - L"expected=%f", - Thread, Row, MatrixThread, UseVectorB ? L"B" : L"A", - static_cast(Actual), static_cast(Expected)); + hlsl_test::LogErrorFmt(L"Thread %d row %d (vector %s): actual=%f, " + L"expected=%f", + Thread, Row, UseVectorB ? L"B" : L"A", + static_cast(Actual), + static_cast(Expected)); Success = false; } else if (Verbose) - hlsl_test::LogCommentFmt(L"Thread %d row %d (matrix %d, vector %s): %f", - Thread, Row, MatrixThread, - UseVectorB ? L"B" : L"A", + hlsl_test::LogCommentFmt(L"Thread %d row %d (vector %s): %f", Thread, + Row, UseVectorB ? L"B" : L"A", static_cast(Actual)); } } @@ -8062,9 +7788,10 @@ verifyThreadSemanticsOutput(const void *Data, size_t Size, bool Divergent, // rather than violated. if (Divergent && MixedWaves == 0) { hlsl_test::LogCommentFmt( - L"No wave reported both input selections, so non-uniform selection " - L"was not exercised. The per-thread results were verified before " - L"reporting this case as inconclusive"); + L"Every thread reported a wave that executed a single arm, so the " + L"thread-scope operations never ran under non-uniform control flow " + L"and this case establishes nothing about it. The per-thread results " + L"were verified before reporting this case as inconclusive"); return ThreadSemanticsOutcome::Inconclusive; } if (Divergent && Verbose) @@ -8158,65 +7885,6 @@ static const char ThreadDivergentMatVecShader[] = R"( } )"; -static const char ThreadMatrixSelectionShader[] = R"( - ByteAddressBuffer MatrixInput : register(t0); - ByteAddressBuffer VectorInput : register(t1); - RWByteAddressBuffer Output : register(u2); - - using Matrix = __builtin_LinAlgMatrix - [[__LinAlgMatrix_Attributes(COMP_TYPE, M_DIM, N_DIM, USE, SCOPE)]]; - - [numthreads(NUMTHREADS, 1, 1)] - void main(uint T : SV_GroupIndex) { - Matrix A, B, Selected; - dx::__builtin_LinAlg_MatrixLoadFromDescriptor( - A, MatrixInput, T * MATRIX_SLOT_BYTES, STRIDE, LAYOUT, - MATRIX_SLOT_BYTES); - dx::__builtin_LinAlg_MatrixLoadFromDescriptor( - B, MatrixInput, (NUMTHREADS - 1 - T) * MATRIX_SLOT_BYTES, STRIDE, - LAYOUT, MATRIX_SLOT_BYTES); - - const bool UseB = (WaveGetLaneIndex() & 1) != 0; - const bool Mixed = WaveActiveAnyTrue(UseB) && WaveActiveAnyTrue(!UseB); - Output.Store(T * OUTPUT_SLOT_BYTES + WITNESS_OFFSET, - (UseB ? WITNESS_USE_B : 0) | (Mixed ? WITNESS_MIXED : 0)); - -#if MATRIX_SELECTION_MODE == 0 - Matrix Choices[2] = {A, B}; - Selected = Choices[UseB ? 1 : 0]; - Output.Store(T * OUTPUT_SLOT_BYTES + ARM_OFFSET, UseB ? 1 : 0); -#elif MATRIX_SELECTION_MODE == 1 - [branch] - if (UseB) { - Selected = B; - Output.Store(T * OUTPUT_SLOT_BYTES + ARM_OFFSET, 1); - } else { - Selected = A; - Output.Store(T * OUTPUT_SLOT_BYTES + ARM_OFFSET, 0); - } -#elif MATRIX_SELECTION_MODE == 2 - [flatten] - if (UseB) - Selected = B; - else - Selected = A; - Output.Store(T * OUTPUT_SLOT_BYTES + ARM_OFFSET, UseB ? 1 : 0); -#else -#error Invalid matrix selection mode -#endif - - vector InVec; - for (uint I = 0; I < N_DIM; ++I) - InVec[I] = VectorInput.Load( - (UseB ? VEC_B_OFFSET : 0) + I * ELEM_SIZE); - vector OutVec; - dx::__builtin_LinAlg_MatrixVectorMultiply( - OutVec, Selected, 1, InVec, IN_INTERP); - for (uint I = 0; I < M_DIM; ++I) - Output.Store(T * OUTPUT_SLOT_BYTES + I * ELEM_SIZE, OutVec[I]); - } -)"; - static MatrixParams makeThreadSemanticsParams() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -8234,8 +7902,7 @@ static void runThreadSemanticsMatVec(ID3D12Device *Device, dxc::SpecificDllLoader &DxcSupport, const MatrixParams &Params, const char *Shader, bool Divergent, - bool Verbose, bool MatrixSelection = false, - const char *ExtraArgs = "") { + bool Verbose) { VERIFY_ARE_EQUAL(Params.strideBytes(), ThreadSemanticsStrideBytes); std::stringstream ExtraDefs; @@ -8247,7 +7914,6 @@ static void runThreadSemanticsMatVec(ID3D12Device *Device, ExtraDefs << " -DARM_OFFSET=" << ThreadSemanticsArmOffset; ExtraDefs << " -DWITNESS_USE_B=" << ThreadSemanticsWitnessUseB; ExtraDefs << " -DWITNESS_MIXED=" << ThreadSemanticsWitnessMixed; - ExtraDefs << ExtraArgs; const std::string Args = buildCompilerArgs(Params, ExtraDefs.str().c_str()); compileShader(DxcSupport, Shader, "cs_6_10", Args, Verbose); @@ -8290,7 +7956,7 @@ static void runThreadSemanticsMatVec(ID3D12Device *Device, MappedData OutData; Result->Test->GetReadBackData("Output", &OutData); switch (verifyThreadSemanticsOutput(OutData.data(), OutData.size(), Divergent, - Verbose, MatrixSelection)) { + Verbose)) { case ThreadSemanticsOutcome::Verified: return; case ThreadSemanticsOutcome::Inconclusive: @@ -8303,7 +7969,7 @@ static void runThreadSemanticsMatVec(ID3D12Device *Device, VERIFY_IS_TRUE(false, "Unknown thread semantics outcome"); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_PerThread() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F16_PerThread() { const MatrixParams Params = makeThreadSemanticsParams(); if (!matVecMulApplicable(D3DDevice, Params, ComponentType::F16, @@ -8317,7 +7983,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_PerThread() { VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_Divergent() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F16_Divergent() { const MatrixParams Params = makeThreadSemanticsParams(); if (!matVecMulApplicable(D3DDevice, Params, ComponentType::F16, @@ -8331,42 +7997,6 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_Divergent() { VerboseLogging); } -static void runThreadMatrixSelection(ID3D12Device *Device, - dxc::SpecificDllLoader &DxcSupport, - UINT Mode, LPCWSTR CaseName, - bool Verbose) { - const MatrixParams Params = makeThreadSemanticsParams(); - if (!matVecMulApplicable(Device, Params, ComponentType::F16, - /*HasBias=*/false, - linalg_test::CapabilityRequirement::Mandatory, - CaseName)) - return; - - const std::string ExtraArgs = - " -DMATRIX_SELECTION_MODE=" + std::to_string(Mode); - runThreadSemanticsMatVec( - Device, DxcSupport, Params, ThreadMatrixSelectionShader, - /*Divergent=*/true, Verbose, /*MatrixSelection=*/true, ExtraArgs.c_str()); -} - -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_MatrixArray() { - runThreadMatrixSelection(D3DDevice, DxcSupport, 0, - L"MatVecMul_Thread_4x8_F16_MatrixArray", - VerboseLogging); -} - -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_MatrixPhi() { - runThreadMatrixSelection(D3DDevice, DxcSupport, 1, - L"MatVecMul_Thread_4x8_F16_MatrixPhi", - VerboseLogging); -} - -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_MatrixSelect() { - runThreadMatrixSelection(D3DDevice, DxcSupport, 2, - L"MatVecMul_Thread_4x8_F16_MatrixSelect", - VerboseLogging); -} - static const char MatVecMulAddShader[] = R"( #define USE_A 0 #define SCOPE_THREAD 0 @@ -8447,7 +8077,7 @@ static void runMatVecMulAdd(ID3D12Device *Device, Expected, Params.M, Verbose)); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMulAdd_Thread_16x16_F16() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -8468,7 +8098,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMulAdd_Thread_16x16_F16() { /*FillValue=*/2, /*OutputSigned=*/true, ComponentType::F16); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMulAdd_Thread_4x8_F32() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_4x8_F32() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -8624,8 +8254,7 @@ static void runOuterProduct(ID3D12Device *Device, } #endif // defined(HLSLEXEC_LINALG_HOST_API) -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: - OuterProduct_Thread_16x16_F16() { +void DxilConf_SM610_LinAlg::OuterProduct_Thread_16x16_F16() { #if defined(HLSLEXEC_LINALG_HOST_API) MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -8818,7 +8447,7 @@ static void runQueryAccumLayout(ID3D12Device *Device, hlsl_test::LogCommentFmt(L"AccumulatorLayout = %u", Layout); } -void DxilConf_SM610_LinAlg_MatrixArithmetic::QueryAccumLayout() { +void DxilConf_SM610_LinAlg::QueryAccumLayout() { if (!linAlgTierApplicable(D3DDevice, L"QueryAccumLayout")) return; @@ -8921,7 +8550,7 @@ static void runLoadMemory(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_GroupSharedIO::LoadMemory_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::LoadMemory_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -9014,7 +8643,7 @@ static void runStoreMemory(ID3D12Device *Device, Expected, NumElements, Verbose)); } -void DxilConf_SM610_LinAlg_GroupSharedIO::StoreMemory_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::StoreMemory_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -9110,7 +8739,7 @@ static void runAccumulateMemory(ID3D12Device *Device, static void runPaddedGroupSharedAccumulateCase( ID3D12Device *Device, dxc::SpecificDllLoader &DxcSupport, bool Verbose); -void DxilConf_SM610_LinAlg_GroupSharedIO::AccumulateMemory_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::AccumulateMemory_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -9591,7 +9220,7 @@ static void runGroupSharedI8Multiply(ID3D12Device *Device, VERIFY_IS_TRUE(AllWavesMatch); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: +void DxilConf_SM610_LinAlg:: MatMatMulAccumMemory_Wave_8x32x16_I8_ToI32_OffsetPadded() { runGroupSharedI8Multiply( D3DDevice, DxcSupport, /*M=*/8, @@ -9599,7 +9228,7 @@ void DxilConf_SM610_LinAlg_GroupSharedIO:: VerboseLogging); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: +void DxilConf_SM610_LinAlg:: MatMatMulAccumMemory_Wave_16x32x16_I8_ToI32_OffsetPadded() { runGroupSharedI8Multiply( D3DDevice, DxcSupport, /*M=*/16, @@ -9779,7 +9408,7 @@ static void runBidirectionalGroupSharedTransfer( GroupSharedLimit); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: +void DxilConf_SM610_LinAlg:: LoadStoreMemory_Wave_4x8_F16_RowMajorOffsetPadded() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -9816,7 +9445,7 @@ void DxilConf_SM610_LinAlg_GroupSharedIO:: } // Keep this rectangular too: both transfer directions must expose layout swaps. -void DxilConf_SM610_LinAlg_GroupSharedIO:: +void DxilConf_SM610_LinAlg:: LoadStoreMemory_Wave_16x32_F16_RowMajorOffsetPadded() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; @@ -9852,7 +9481,7 @@ void DxilConf_SM610_LinAlg_GroupSharedIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: +void DxilConf_SM610_LinAlg:: LoadStoreMemory_Wave_4x8_F32_ColumnMajorOffsetPadded() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; @@ -9888,8 +9517,7 @@ void DxilConf_SM610_LinAlg_GroupSharedIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: - LoadStoreMemory_ThreadGroup_4x8_F16() { +void DxilConf_SM610_LinAlg::LoadStoreMemory_ThreadGroup_4x8_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 4; @@ -9923,8 +9551,7 @@ void DxilConf_SM610_LinAlg_GroupSharedIO:: SelectedWaveSize); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: - LoadStoreMemory_ThreadGroup_WaveScaled_F16() { +void DxilConf_SM610_LinAlg::LoadStoreMemory_ThreadGroup_WaveScaled_F16() { const LPCWSTR CaseName = L"LoadStoreMemory_ThreadGroup_WaveScaled_F16"; if (!linAlgTierApplicable(D3DDevice, CaseName)) return; @@ -10329,15 +9956,13 @@ static void runGroupSharedAccumulateContention( SelectedWaveSize, ActiveWaveCount); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: - AccumulateMemoryContention_Wave_4x8_F16() { +void DxilConf_SM610_LinAlg::AccumulateMemoryContention_Wave_4x8_F16() { runGroupSharedAccumulateContention(D3DDevice, DxcSupport, ComponentType::F16, L"AccumulateMemoryContention_Wave_4x8_F16", VerboseLogging); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: - AccumulateMemoryContention_Wave_16x16_F16() { +void DxilConf_SM610_LinAlg::AccumulateMemoryContention_Wave_16x16_F16() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 16; @@ -10373,8 +9998,7 @@ void DxilConf_SM610_LinAlg_GroupSharedIO:: VerboseLogging, SelectedWaveSize, ActiveWaveCount); } -void DxilConf_SM610_LinAlg_GroupSharedIO:: - AccumulateMemoryContention_Wave_4x8_I32() { +void DxilConf_SM610_LinAlg::AccumulateMemoryContention_Wave_4x8_I32() { runGroupSharedAccumulateContention(D3DDevice, DxcSupport, ComponentType::I32, L"AccumulateMemoryContention_Wave_4x8_I32", VerboseLogging); @@ -10421,16 +10045,10 @@ static void runConvert(ID3D12Device *Device, dxc::SpecificDllLoader &DxcSupport, Expected, NumElements, Verbose)); } -static bool -convertTypesApplicable(ID3D12Device *Device, ComponentType SourceCompType, - ComponentType DestinationCompType, - linalg_test::CapabilityRequirement Requirement, - LPCWSTR CaseName); - -void DxilConf_SM610_LinAlg_Conversion::Convert() { - if (!convertTypesApplicable( - D3DDevice, ComponentType::F16, ComponentType::F32, - linalg_test::CapabilityRequirement::CapabilityGated, L"Convert")) +void DxilConf_SM610_LinAlg::Convert() { + // Operates on vectors rather than matrices, so tier support is the only + // capability it needs. + if (!linAlgTierApplicable(D3DDevice, L"Convert")) return; runConvert(D3DDevice, DxcSupport, VerboseLogging); @@ -10599,8 +10217,7 @@ static void runVectorAccumulateDescriptor( OutData.size(), Verbose)); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: - VectorAccumulateDescriptor_Thread_F16() { +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptor_Thread_F16() { // Tier 1 requires no accumulation store formats, so this is gated. if (!accumulateStoreApplicable( D3DDevice, ComponentType::F16, @@ -10634,7 +10251,7 @@ void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: VerboseLogging); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: +void DxilConf_SM610_LinAlg:: VectorAccumulateDescriptor_Thread_F16_Length8_NonZero() { if (!accumulateStoreApplicable( D3DDevice, ComponentType::F16, @@ -10676,7 +10293,7 @@ void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: VerboseLogging); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: +void DxilConf_SM610_LinAlg:: VectorAccumulateDescriptor_Thread_F32_Length8_NonZero() { if (!accumulateStoreApplicable( D3DDevice, ComponentType::F32, @@ -10765,15 +10382,13 @@ runVectorAccumulateDescriptorOutOfBounds(ID3D12Device *Device, /*OutputViewBytes=*/StartOffsetBytes); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: - VectorAccumulateDescriptorOOB_Thread_F16() { +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorOOB_Thread_F16() { runVectorAccumulateDescriptorOutOfBounds( D3DDevice, DxcSupport, L"VectorAccumulateDescriptorOOB_Thread_F16", VerboseLogging); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: - VectorAccumulateDescriptorOOB_Thread_F32() { +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorOOB_Thread_F32() { runVectorAccumulateDescriptorOutOfBounds( D3DDevice, DxcSupport, L"VectorAccumulateDescriptorOOB_Thread_F32", VerboseLogging); @@ -10799,8 +10414,7 @@ static_assert(VectorContentionInvocations == "of the invocation index, so the expected sums hold only when " "every digit combination occurs exactly once"); -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: - VectorAccumulateDescriptorContention_Thread_F16() { +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorContention_Thread_F16() { if (!accumulateStoreApplicable( D3DDevice, ComponentType::F16, linalg_test::AtomicDestination::RWByteAddressBuffer, @@ -10844,7 +10458,7 @@ void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: VerboseLogging, VectorContentionThreads, VectorContentionGroups); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: +void DxilConf_SM610_LinAlg:: VectorAccumulateDescriptorContention_Thread_F32_OrderInvariant() { if (!accumulateStoreApplicable( D3DDevice, ComponentType::F32, @@ -10882,8 +10496,7 @@ void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: VerboseLogging, VectorContentionThreads, VectorContentionGroups); } -void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: - VectorAccumulateDescriptorContention_Thread_I32() { +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorContention_Thread_I32() { if (!accumulateStoreApplicable( D3DDevice, ComponentType::I32, linalg_test::AtomicDestination::RWByteAddressBuffer, @@ -10912,7 +10525,7 @@ void DxilConf_SM610_LinAlg_OuterVectorAccumulation:: VerboseLogging, VectorContentionThreads, VectorContentionGroups); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_NonUniform() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F16_NonUniform() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeNonUniformF16Case(MatrixLayout::RowMajor); matvec_interpretation::runCapabilityChecked( @@ -10921,7 +10534,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_NonUniform() { L"MatVecMul_Thread_4x8_F16_NonUniform", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_ColumnMajor() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F16_ColumnMajor() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeNonUniformF16Case(MatrixLayout::ColumnMajor); matvec_interpretation::runCapabilityChecked( @@ -10930,7 +10543,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F16_ColumnMajor() { L"MatVecMul_Thread_4x8_F16_ColumnMajor", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_I8_Interpreted() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_I8_Interpreted() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeSInt8Case(); matvec_interpretation::runCapabilityChecked( @@ -10939,7 +10552,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_I8_Interpreted() { L"MatVecMul_Thread_4x8_I8_Interpreted", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_U8_Interpreted() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_U8_Interpreted() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeUInt8Case(); matvec_interpretation::runCapabilityChecked( @@ -10948,7 +10561,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_U8_Interpreted() { L"MatVecMul_Thread_4x8_U8_Interpreted", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_U32_UnsignedOutput() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_U32_UnsignedOutput() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeUInt32OutputCase(); matvec_interpretation::runCapabilityChecked( @@ -10957,7 +10570,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_U32_UnsignedOutput() { L"MatVecMul_Thread_4x8_U32_UnsignedOutput", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x16_F8_E4M3FN() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x16_F8_E4M3FN() { #if defined(HLSLEXEC_LINALG_HOST_API) const matvec_interpretation::CaseData Case = matvec_interpretation::makeFP8MatrixCase(ComponentType::F8_E4M3FN); @@ -10971,7 +10584,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x16_F8_E4M3FN() { #endif // defined(HLSLEXEC_LINALG_HOST_API) } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x16_F8_E5M2() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x16_F8_E5M2() { #if defined(HLSLEXEC_LINALG_HOST_API) const matvec_interpretation::CaseData Case = matvec_interpretation::makeFP8MatrixCase(ComponentType::F8_E5M2); @@ -10985,8 +10598,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x16_F8_E5M2() { #endif // defined(HLSLEXEC_LINALG_HOST_API) } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMul_Thread_4x16_F8_E4M3FN_MatchedInputs() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x16_F8_E4M3FN_MatchedInputs() { #if defined(HLSLEXEC_LINALG_HOST_API) const matvec_interpretation::CaseData Case = matvec_interpretation::makeMatchedFP8Case(ComponentType::F8_E4M3FN, @@ -11001,8 +10613,7 @@ void DxilConf_SM610_LinAlg_MatVec:: #endif // defined(HLSLEXEC_LINALG_HOST_API) } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMul_Thread_4x16_F8_E5M2_MatchedInputs() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x16_F8_E5M2_MatchedInputs() { #if defined(HLSLEXEC_LINALG_HOST_API) const matvec_interpretation::CaseData Case = matvec_interpretation::makeMatchedFP8Case(ComponentType::F8_E5M2, @@ -11017,7 +10628,7 @@ void DxilConf_SM610_LinAlg_MatVec:: #endif // defined(HLSLEXEC_LINALG_HOST_API) } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F8_E4M3FN_Vector() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F8_E4M3FN_Vector() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeFP8VectorCase(ComponentType::F8_E4M3FN); matvec_interpretation::runCapabilityChecked( @@ -11026,7 +10637,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F8_E4M3FN_Vector() { L"MatVecMul_Thread_4x8_F8_E4M3FN_Vector", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F8_E5M2_Vector() { +void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F8_E5M2_Vector() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeFP8VectorCase(ComponentType::F8_E5M2); matvec_interpretation::runCapabilityChecked( @@ -11035,8 +10646,7 @@ void DxilConf_SM610_LinAlg_MatVec::MatVecMul_Thread_4x8_F8_E5M2_Vector() { L"MatVecMul_Thread_4x8_F8_E5M2_Vector", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMulAdd_Thread_4x16_F8_E4M3FN_MatchedInputs() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_4x16_F8_E4M3FN_MatchedInputs() { #if defined(HLSLEXEC_LINALG_HOST_API) const matvec_interpretation::CaseData Case = matvec_interpretation::makeMatchedFP8Case(ComponentType::F8_E4M3FN, @@ -11051,8 +10661,7 @@ void DxilConf_SM610_LinAlg_MatVec:: #endif // defined(HLSLEXEC_LINALG_HOST_API) } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMulAdd_Thread_4x16_F8_E5M2_MatchedInputs() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_4x16_F8_E5M2_MatchedInputs() { #if defined(HLSLEXEC_LINALG_HOST_API) const matvec_interpretation::CaseData Case = matvec_interpretation::makeMatchedFP8Case(ComponentType::F8_E5M2, @@ -11067,8 +10676,7 @@ void DxilConf_SM610_LinAlg_MatVec:: #endif // defined(HLSLEXEC_LINALG_HOST_API) } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMulAdd_Thread_4x8_F8_E4M3FN_MemoryBias() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_4x8_F8_E4M3FN_MemoryBias() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeFP8MemoryBiasCase(ComponentType::F8_E4M3FN); matvec_interpretation::runCapabilityChecked( @@ -11077,8 +10685,7 @@ void DxilConf_SM610_LinAlg_MatVec:: L"MatVecMulAdd_Thread_4x8_F8_E4M3FN_MemoryBias", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMulAdd_Thread_4x8_F8_E5M2_MemoryBias() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_4x8_F8_E5M2_MemoryBias() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeFP8MemoryBiasCase(ComponentType::F8_E5M2); matvec_interpretation::runCapabilityChecked( @@ -11087,8 +10694,7 @@ void DxilConf_SM610_LinAlg_MatVec:: L"MatVecMulAdd_Thread_4x8_F8_E5M2_MemoryBias", VerboseLogging); } -void DxilConf_SM610_LinAlg_MatVec:: - MatVecMulAdd_Thread_4x8_F16_IndependentBias() { +void DxilConf_SM610_LinAlg::MatVecMulAdd_Thread_4x8_F16_IndependentBias() { matvec_interpretation::CaseData Case = matvec_interpretation::makeNonUniformF16Case(MatrixLayout::RowMajor); Case.BiasInputType = ComponentType::F16; @@ -11465,17 +11071,17 @@ static const char ConvertU32ToU16CoverageShader[] = R"( )"; static const char ConvertF32ToI16CoverageShader[] = R"( - ByteAddressBuffer Input : register(t0); RWByteAddressBuffer Output : register(u0); [numthreads(1, 1, 1)] void main() { - vector InVec; - for (uint I = 0; I < NUM_ELEMENTS; ++I) - InVec[I] = Input.Load(I * 4); - vector OutVec; + vector InVec = { + -40000.0F, -32768.5F, -2.5F, -1.5F, + 1.5F, 2.5F, 32767.5F, 40000.0F + }; + vector OutVec; dx::__builtin_LinAlg_Convert(OutVec, InVec, SRC_TYPE, DST_TYPE); - for (uint I = 0; I < NUM_ELEMENTS; ++I) + for (uint I = 0; I < 8; ++I) Output.Store(I * 2, OutVec[I]); } )"; @@ -11500,7 +11106,7 @@ static const char ConvertI32ToF16CoverageShader[] = R"( )"; static const char ConvertF16FP8CoverageShader[] = R"( - ByteAddressBuffer Input : register(t0); + ByteAddressBuffer DecodeInput : register(t0); RWByteAddressBuffer Output : register(u0); [numthreads(1, 1, 1)] @@ -11512,7 +11118,7 @@ static const char ConvertF16FP8CoverageShader[] = R"( dx::__builtin_LinAlg_Convert(Packed, EncodeInput, SRC_TYPE, DST_TYPE); vector HostPacked = { - Input.Load(0), Input.Load(4) + DecodeInput.Load(0), DecodeInput.Load(4) }; vector Decoded; dx::__builtin_LinAlg_Convert(Decoded, HostPacked, DST_TYPE, SRC_TYPE); @@ -11524,12 +11130,12 @@ static const char ConvertF16FP8CoverageShader[] = R"( } )"; -template -static void -runConvertCoverage(ID3D12Device *Device, dxc::SpecificDllLoader &DxcSupport, - const char *Shader, const std::string &Args, - size_t OutputBytes, Verifier Verify, bool Verbose, - const std::vector *Input = nullptr) { +static void runExactConvert(ID3D12Device *Device, + dxc::SpecificDllLoader &DxcSupport, + const char *Shader, const std::string &Args, + const std::vector &Expected, + LPCWSTR PublicRule, bool Verbose, + const std::vector *Input = nullptr) { compileShader(DxcSupport, Shader, "cs_6_10", Args, Verbose); auto Op = createComputeOp( @@ -11539,41 +11145,24 @@ runConvertCoverage(ID3D12Device *Device, dxc::SpecificDllLoader &DxcSupport, VERIFY_IS_TRUE(!Input->empty(), "Convert input must not be empty"); if (Input->empty()) return; - addSRVBuffer(Op.get(), "Input", Input->size(), "byname"); - addRootView(Op.get(), 0, "Input"); + addSRVBuffer(Op.get(), "DecodeInput", Input->size(), "byname"); + addRootView(Op.get(), 0, "DecodeInput"); OutputRootIndex = 1; } - addUAVBuffer(Op.get(), "Output", OutputBytes, true, "byname"); + addUAVBuffer(Op.get(), "Output", Expected.size(), true); addRootView(Op.get(), OutputRootIndex, "Output"); auto Result = runShaderOp( Device, DxcSupport, std::move(Op), [Input](LPCSTR Name, std::vector &Data, st::ShaderOp *) { - if (_stricmp(Name, "Output") == 0) - cpu_oracle::fillPoison(Data.data(), Data.size()); - else if (Input && _stricmp(Name, "Input") == 0) + if (Input && _stricmp(Name, "DecodeInput") == 0) Data = *Input; - else - VERIFY_IS_TRUE(false, "Unexpected conversion resource initializer"); }); MappedData OutputData; Result->Test->GetReadBackData("Output", &OutputData); - VERIFY_IS_TRUE(Verify(OutputData.data(), OutputData.size())); -} - -static void runExactConvert(ID3D12Device *Device, - dxc::SpecificDllLoader &DxcSupport, - const char *Shader, const std::string &Args, - const std::vector &Expected, - LPCWSTR PublicRule, bool Verbose, - const std::vector *Input = nullptr) { - runConvertCoverage( - Device, DxcSupport, Shader, Args, Expected.size(), - [&](const void *Data, size_t Size) { - return verifyConvertBytes(Data, Size, Expected, PublicRule, Verbose); - }, - Verbose, Input); + VERIFY_IS_TRUE(verifyConvertBytes(OutputData.data(), OutputData.size(), + Expected, PublicRule, Verbose)); } struct FP8ConvertData { @@ -11742,7 +11331,7 @@ static std::string buildConvertArgs(ComponentType SourceCompType, return Args.str(); } -void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_4x8_F16_ToF32() { +void DxilConf_SM610_LinAlg::CopyConvert_Wave_4x8_F16_ToF32() { MatrixParams Params = {}; Params.CompType = ComponentType::F16; Params.M = 4; @@ -11763,8 +11352,7 @@ void DxilConf_SM610_LinAlg_Conversion::CopyConvert_Wave_4x8_F16_ToF32() { VerboseLogging, /*Transpose=*/false, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_Conversion:: - CopyConvert_Wave_4x8_F32_ToF16_Transpose() { +void DxilConf_SM610_LinAlg::CopyConvert_Wave_4x8_F32_ToF16_Transpose() { MatrixParams Params = {}; Params.CompType = ComponentType::F32; Params.M = 4; @@ -11785,7 +11373,7 @@ void DxilConf_SM610_LinAlg_Conversion:: VerboseLogging, /*Transpose=*/true, SelectedWaveSize); } -void DxilConf_SM610_LinAlg_Conversion::Convert_I16_ToI32_Exact() { +void DxilConf_SM610_LinAlg::Convert_I16_ToI32_Exact() { if (!convertTypesApplicable( D3DDevice, ComponentType::I16, ComponentType::I32, linalg_test::CapabilityRequirement::CapabilityGated, @@ -11801,7 +11389,7 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_I16_ToI32_Exact() { VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_I32_ToI16_Saturate() { +void DxilConf_SM610_LinAlg::Convert_I32_ToI16_Saturate() { if (!convertTypesApplicable( D3DDevice, ComponentType::I32, ComponentType::I16, linalg_test::CapabilityRequirement::CapabilityGated, @@ -11816,7 +11404,7 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_I32_ToI16_Saturate() { VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_I32_ToU16_Saturate() { +void DxilConf_SM610_LinAlg::Convert_I32_ToU16_Saturate() { if (!convertTypesApplicable( D3DDevice, ComponentType::I32, ComponentType::U16, linalg_test::CapabilityRequirement::CapabilityGated, @@ -11831,7 +11419,7 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_I32_ToU16_Saturate() { VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_U32_ToI16_Saturate() { +void DxilConf_SM610_LinAlg::Convert_U32_ToI16_Saturate() { if (!convertTypesApplicable( D3DDevice, ComponentType::U32, ComponentType::I16, linalg_test::CapabilityRequirement::CapabilityGated, @@ -11847,7 +11435,7 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_U32_ToI16_Saturate() { VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_U32_ToU16_Saturate() { +void DxilConf_SM610_LinAlg::Convert_U32_ToU16_Saturate() { if (!convertTypesApplicable( D3DDevice, ComponentType::U32, ComponentType::U16, linalg_test::CapabilityRequirement::CapabilityGated, @@ -11875,74 +11463,26 @@ static void runFP8ConvertCase(ID3D12Device *Device, &Data.DecodeInput); } -static void runF32ToI16Convert(ID3D12Device *Device, - dxc::SpecificDllLoader &DxcSupport, - bool NonFinite, LPCWSTR CaseName, bool Verbose) { +void DxilConf_SM610_LinAlg::Convert_F32_ToI16_RTNE_Saturate() { if (!convertTypesApplicable( - Device, ComponentType::F32, ComponentType::I16, - linalg_test::CapabilityRequirement::CapabilityGated, CaseName)) - return; - - std::vector Cases; - for (const cpu_oracle::FloatToIntCase &Case : cpu_oracle::F32ToI16Cases) { - const bool IsNonFinite = !std::isfinite(Case.Input); - if (IsNonFinite == NonFinite) - Cases.push_back(Case); - } - const size_t Count = Cases.size(); - std::vector Input(Count * sizeof(float)); - for (size_t I = 0; I < Count; ++I) - std::memcpy(Input.data() + I * sizeof(float), &Cases[I].Input, - sizeof(float)); - const size_t OutputBytes = Count * sizeof(int16_t); - const std::string Args = - buildConvertArgs(ComponentType::F32, ComponentType::I16) + - " -DNUM_ELEMENTS=" + std::to_string(Count); - runConvertCoverage( - Device, DxcSupport, ConvertF32ToI16CoverageShader, Args, OutputBytes, - [&Cases, OutputBytes](const void *Data, size_t Size) { - if (Size != OutputBytes) { - hlsl_test::LogErrorFmt( - L"Float-to-I16 output size: actual=%zu, expected=%zu", Size, - OutputBytes); - return false; - } - const BYTE *Bytes = static_cast(Data); - bool Success = true; - for (size_t I = 0; I < Cases.size(); ++I) { - const cpu_oracle::FloatToIntCase &Case = Cases[I]; - int16_t Actual; - std::memcpy(&Actual, Bytes + I * sizeof(Actual), sizeof(Actual)); - if (!cpu_oracle::isNearestSaturatedI16(Case.Input, Actual)) { - hlsl_test::LogErrorFmt( - L"Float-to-I16 element %zu: input=%g, actual=%d, " - L"permitted=[%d,%d]", - I, static_cast(Case.Input), Actual, Case.Lower, - Case.Upper); - Success = false; - } - } - return Success; - }, - Verbose, &Input); -} - -void DxilConf_SM610_LinAlg_Conversion:: - Convert_F32_ToI16_RoundNearest_Saturate() { - runF32ToI16Convert(D3DDevice, DxcSupport, /*NonFinite=*/false, - L"Convert_F32_ToI16_RoundNearest_Saturate", - VerboseLogging); -} + D3DDevice, ComponentType::F32, ComponentType::I16, + linalg_test::CapabilityRequirement::CapabilityGated, + L"Convert_F32_ToI16_RTNE_Saturate")) + return; -void DxilConf_SM610_LinAlg_Conversion::Convert_F32_ToI16_NonFinite() { - runF32ToI16Convert(D3DDevice, DxcSupport, /*NonFinite=*/true, - L"Convert_F32_ToI16_NonFinite", VerboseLogging); + runExactConvert(D3DDevice, DxcSupport, ConvertF32ToI16CoverageShader, + buildConvertArgs(ComponentType::F32, ComponentType::I16), + encodeConvertVector( + {-32768, -32768, -2, -2, 2, 2, 32767, 32767}), + L"Float-to-integer conversion is RTNE with signed saturation", + VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_I32_ToF16_RTNE() { - if (!convertTypesApplicable(D3DDevice, ComponentType::I32, ComponentType::F16, - linalg_test::CapabilityRequirement::Mandatory, - L"Convert_I32_ToF16_RTNE")) +void DxilConf_SM610_LinAlg::Convert_I32_ToF16_RTNE() { + if (!convertTypesApplicable( + D3DDevice, ComponentType::I32, ComponentType::F16, + linalg_test::CapabilityRequirement::CapabilityGated, + L"Convert_I32_ToF16_RTNE")) return; // Hand-derived F16 bit patterns: 2048 and 2052 are 0x6800 and 0x6802, 4096 @@ -11955,7 +11495,7 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_I32_ToF16_RTNE() { L"Integer-to-float conversion is RTNE", VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_F16_ToE4M3FN_AndBack() { +void DxilConf_SM610_LinAlg::Convert_F16_ToE4M3FN_AndBack() { const std::optional Data = makeFP8ConvertData(ComponentType::F8_E4M3FN); VERIFY_IS_TRUE(Data.has_value(), "Unable to construct the host FP8 oracle"); @@ -11970,15 +11510,16 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_F16_ToE4M3FN_AndBack() { linalg_test::CapabilityRequirement::Mandatory, L"Convert_F16_ToE4M3FN_AndBack")) return; - if (!convertTypesApplicable(D3DDevice, ComponentType::U32, ComponentType::F16, - linalg_test::CapabilityRequirement::Mandatory, - L"Convert_F16_ToE4M3FN_AndBack")) + if (!convertTypesApplicable( + D3DDevice, ComponentType::U32, ComponentType::F16, + linalg_test::CapabilityRequirement::CapabilityGated, + L"Convert_F16_ToE4M3FN_AndBack")) return; runFP8ConvertCase(D3DDevice, DxcSupport, ComponentType::F8_E4M3FN, *Data, VerboseLogging); } -void DxilConf_SM610_LinAlg_Conversion::Convert_F16_ToE5M2_AndBack() { +void DxilConf_SM610_LinAlg::Convert_F16_ToE5M2_AndBack() { const std::optional Data = makeFP8ConvertData(ComponentType::F8_E5M2); VERIFY_IS_TRUE(Data.has_value(), "Unable to construct the host FP8 oracle"); @@ -11993,9 +11534,10 @@ void DxilConf_SM610_LinAlg_Conversion::Convert_F16_ToE5M2_AndBack() { linalg_test::CapabilityRequirement::Mandatory, L"Convert_F16_ToE5M2_AndBack")) return; - if (!convertTypesApplicable(D3DDevice, ComponentType::U32, ComponentType::F16, - linalg_test::CapabilityRequirement::Mandatory, - L"Convert_F16_ToE5M2_AndBack")) + if (!convertTypesApplicable( + D3DDevice, ComponentType::U32, ComponentType::F16, + linalg_test::CapabilityRequirement::CapabilityGated, + L"Convert_F16_ToE5M2_AndBack")) return; runFP8ConvertCase(D3DDevice, DxcSupport, ComponentType::F8_E5M2, *Data, VerboseLogging); diff --git a/tools/clang/unittests/Lex/CMakeLists.txt b/tools/clang/unittests/Lex/CMakeLists.txt index 12c307ad82..461e0d95fc 100644 --- a/tools/clang/unittests/Lex/CMakeLists.txt +++ b/tools/clang/unittests/Lex/CMakeLists.txt @@ -6,7 +6,6 @@ add_clang_unittest(LexTests LexerTest.cpp PPCallbacksTest.cpp PPConditionalDirectiveRecordTest.cpp - PPOptionsTests.cpp ) target_link_libraries(LexTests diff --git a/tools/clang/unittests/Lex/PPOptionsTests.cpp b/tools/clang/unittests/Lex/PPOptionsTests.cpp deleted file mode 100644 index 3d886e00ec..0000000000 --- a/tools/clang/unittests/Lex/PPOptionsTests.cpp +++ /dev/null @@ -1,27 +0,0 @@ -/////////////////////////////////////////////////////////////////////////////// -// // -// PPOptionsTests.cpp // -// This file is distributed under the University of Illinois Open Source // -// License. See LICENSE.TXT for details. // -// // -// Tests HLSL-specific preprocessor option defaults. // -// // -/////////////////////////////////////////////////////////////////////////////// - -#include "clang/Lex/PreprocessorOptions.h" -#include "gtest/gtest.h" - -#include -#include - -using namespace clang; - -TEST(PPOptionsTests, ExpandTokPastingArgDefaultsToFalse) { - alignas( - PreprocessorOptions) unsigned char Storage[sizeof(PreprocessorOptions)]; - std::memset(Storage, 0xff, sizeof(Storage)); - - const PreprocessorOptions *Options = new (Storage) PreprocessorOptions(); - EXPECT_FALSE(Options->ExpandTokPastingArg); - Options->~PreprocessorOptions(); -} \ No newline at end of file diff --git a/utils/hct/gen_intrin_main.txt b/utils/hct/gen_intrin_main.txt index e3cd70bd56..65ab9cdf2d 100644 --- a/utils/hct/gen_intrin_main.txt +++ b/utils/hct/gen_intrin_main.txt @@ -1059,11 +1059,11 @@ void [[]] WriteSamplerFeedbackLevel(in Texture2DArray t, in sampler s, in float< namespace RayQueryMethods { -void [[mutable]] TraceRayInline(in acceleration_struct AccelerationStructure, in uint RayFlags, in uint InstanceInclusionMask, in ray_desc Ray); -bool [[mutable]] Proceed(); -void [[mutable]] Abort(); -void [[mutable]] CommitNonOpaqueTriangleHit(); -void [[mutable]] CommitProceduralPrimitiveHit(in float t); +void [[]] TraceRayInline(in acceleration_struct AccelerationStructure, in uint RayFlags, in uint InstanceInclusionMask, in ray_desc Ray); +bool [[]] Proceed(); +void [[]] Abort(); +void [[]] CommitNonOpaqueTriangleHit(); +void [[]] CommitProceduralPrimitiveHit(in float t); uint [[ro]] CommittedStatus(); uint [[ro]] CandidateType(); float<3,4> [[ro]] CandidateObjectToWorld3x4(); @@ -1135,7 +1135,7 @@ namespace DxHitObjectMethods { uint [[rn,class_prefix,min_sm=6.9]] GetHitKind(); uint [[rn,class_prefix,min_sm=6.9]] GetShaderTableIndex(); void [[class_prefix,min_sm=6.9]] GetAttributes(out udt Attributes); - void [[mutable,class_prefix,min_sm=6.9]] SetShaderTableIndex(in uint RecordIndex); + void [[class_prefix,min_sm=6.9]] SetShaderTableIndex(in uint RecordIndex); uint [[ro,class_prefix,min_sm=6.9]] LoadLocalRootTableConstant(in uint RootConstantOffsetInBytes); uint [[rn,class_prefix,min_sm=6.10]] GetClusterID(); triangle_positions [[rn,class_prefix,min_sm=6.10]] TriangleObjectPositions(); diff --git a/utils/hct/hctdb.py b/utils/hct/hctdb.py index 4eccca941b..a57122b21f 100644 --- a/utils/hct/hctdb.py +++ b/utils/hct/hctdb.py @@ -9454,7 +9454,6 @@ def __init__( max_shader_model, static_member, class_prefix, - mutable_method, ): self.name = name # Function name self.idx = idx # Unique number within namespace @@ -9508,7 +9507,6 @@ def __init__( max_shader_model[1] & 0x0F ) self.static_member = static_member # HLSL static member function - self.mutable_method = mutable_method # Mutates the object; not callable on const self.key = ( ("%3d" % ns_idx) + "!" @@ -9911,7 +9909,6 @@ def process_attr(attr): readnone = False # Not read memory argmemonly = False # Only reads memory through pointer arguments static_member = False # Static member function - mutable_method = False # Mutates the object; not callable on const is_wave = False class_prefix = False # Insert class name as enum_prefix # Is wave-sensitive @@ -9943,9 +9940,6 @@ def process_attr(attr): if a == "static": static_member = True continue - if a == "mutable": - mutable_method = True - continue if a == "class_prefix": class_prefix = True continue @@ -10013,7 +10007,6 @@ def process_attr(attr): max_shader_model, static_member, class_prefix, - mutable_method, ) current_namespace = None @@ -10065,7 +10058,6 @@ def process_attr(attr): max_shader_model, static_member, class_prefix, - mutable_method, ) = process_attr(attr) # Add an entry for this intrinsic. if bracket_cleanup_re.search(opts): @@ -10110,7 +10102,6 @@ def process_attr(attr): max_shader_model, static_member, class_prefix, - mutable_method, ) ) num_entries += 1 diff --git a/utils/hct/hctdb_instrhelp.py b/utils/hct/hctdb_instrhelp.py index 43a569fd2a..af5f7c5484 100644 --- a/utils/hct/hctdb_instrhelp.py +++ b/utils/hct/hctdb_instrhelp.py @@ -1134,8 +1134,6 @@ def get_hlsl_intrinsics(): flags.append("INTRIN_FLAG_IS_WAVE") if i.static_member: flags.append("INTRIN_FLAG_STATIC_MEMBER") - if i.mutable_method: - flags.append("INTRIN_FLAG_MUTABLE_METHOD") if flags: flags = " | ".join(flags) else: