From 4648e0ef1719f557f437d0fd4069e00351b9b241 Mon Sep 17 00:00:00 2001 From: Damyan Pepper Date: Wed, 7 Oct 2026 11:14:56 -0700 Subject: [PATCH] [PIX] Select callable and node shader debug invocations Instrument callable shaders, selecting invocations by dispatchRaysIndex, and select node shader invocations according to launch type, reporting whether the selection is exact or approximate. Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> Copilot-Session: 40dc9de3-617e-4caf-ab0d-fba0a033ed93 --- .../DxilDebugInstrumentation.cpp | 76 +++++- .../pix/DebugCallableShader.hlsl | 38 +++ .../pix/DebugNodeBroadcasting.hlsl | 35 +++ .../pix/DebugNodeCoalescing.hlsl | 36 +++ .../pix/DebugNodeThreadLaunch.hlsl | 30 +++ tools/clang/unittests/HLSL/PixTest.cpp | 226 ++++++++++++++++++ 6 files changed, 437 insertions(+), 4 deletions(-) create mode 100644 tools/clang/test/HLSLFileCheck/pix/DebugCallableShader.hlsl create mode 100644 tools/clang/test/HLSLFileCheck/pix/DebugNodeBroadcasting.hlsl create mode 100644 tools/clang/test/HLSLFileCheck/pix/DebugNodeCoalescing.hlsl create mode 100644 tools/clang/test/HLSLFileCheck/pix/DebugNodeThreadLaunch.hlsl diff --git a/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp b/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp index b7d37bf1dd..62eef73840 100644 --- a/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp +++ b/lib/DxilPIXPasses/DxilDebugInstrumentation.cpp @@ -336,7 +336,10 @@ class DxilDebugInstrumentation : public ModulePass { DXIL::ShaderKind shaderKind); Value *addPixelShaderProlog(BuilderContext &BC, SystemValueIndices SVIndices); Value *addGeometryShaderProlog(BuilderContext &BC); + Value *addThreadIdComparisonProlog(BuilderContext &BC, + DXIL::OpCode threadIdOpCode); Value *addDispatchedShaderProlog(BuilderContext &BC); + Value *addNodeShaderProlog(BuilderContext &BC); Value *addRaygenShaderProlog(BuilderContext &BC); Value *addVertexShaderProlog(BuilderContext &BC, SystemValueIndices SVIndices); @@ -392,6 +395,7 @@ static bool IsInstrumentableShaderKind(DXIL::ShaderKind shaderKind) { case DXIL::ShaderKind::AnyHit: case DXIL::ShaderKind::ClosestHit: case DXIL::ShaderKind::Miss: + case DXIL::ShaderKind::Callable: case DXIL::ShaderKind::Node: return true; default: @@ -512,6 +516,7 @@ DxilDebugInstrumentation::addRequiredSystemValues(BuilderContext &BC, case DXIL::ShaderKind::AnyHit: case DXIL::ShaderKind::ClosestHit: case DXIL::ShaderKind::Miss: + case DXIL::ShaderKind::Callable: case DXIL::ShaderKind::Node: // Dispatch* thread Id is not in the input signature break; @@ -563,14 +568,15 @@ DxilDebugInstrumentation::addRequiredSystemValues(BuilderContext &BC, return SVIndices; } -Value *DxilDebugInstrumentation::addDispatchedShaderProlog(BuilderContext &BC) { +Value *DxilDebugInstrumentation::addThreadIdComparisonProlog( + BuilderContext &BC, DXIL::OpCode threadIdOpCode) { Constant *Zero32Arg = BC.HlslOP->GetU32Const(0); Constant *One32Arg = BC.HlslOP->GetU32Const(1); Constant *Two32Arg = BC.HlslOP->GetU32Const(2); auto ThreadIdFunc = - BC.HlslOP->GetOpFunc(DXIL::OpCode::ThreadId, Type::getInt32Ty(BC.Ctx)); - Constant *Opcode = BC.HlslOP->GetU32Const((unsigned)DXIL::OpCode::ThreadId); + BC.HlslOP->GetOpFunc(threadIdOpCode, Type::getInt32Ty(BC.Ctx)); + Constant *Opcode = BC.HlslOP->GetU32Const((unsigned)threadIdOpCode); auto ThreadIdX = BC.Builder.CreateCall(ThreadIdFunc, {Opcode, Zero32Arg}, "ThreadIdX"); auto ThreadIdY = @@ -598,6 +604,62 @@ Value *DxilDebugInstrumentation::addDispatchedShaderProlog(BuilderContext &BC) { return CompareAll; } +Value *DxilDebugInstrumentation::addDispatchedShaderProlog(BuilderContext &BC) { + return addThreadIdComparisonProlog(BC, DXIL::OpCode::ThreadId); +} + +// The identity of a node shader invocation depends on how the node launches, +// and the three launch types offer different system values. The validator +// allows SV_DispatchThreadID and SV_GroupID only to a broadcasting node, and +// SV_GroupThreadID and SV_GroupIndex only to a broadcasting or a coalescing +// node. A thread launch node gets none of them. +// +// - A broadcasting node runs over a grid, so SV_DispatchThreadID names an +// invocation exactly as it does for a compute shader. +// - A coalescing node has only the position of a thread within its group. +// That narrows the selection to the group size instead of to a single +// invocation. +// - A thread launch node has no thread identity at all, so every invocation +// stays of interest. +// +// The pass report names which of the three applies, so the caller presents the +// selection as exact or approximate. +Value *DxilDebugInstrumentation::addNodeShaderProlog(BuilderContext &BC) { + llvm::Function *function = BC.Builder.GetInsertBlock()->getParent(); + + if (!BC.DM.HasDxilFunctionProps(function)) { + *OSOverride << "NodeInvocationSelection:None\n"; + return BC.HlslOP->GetI1Const(1); + } + + hlsl::DxilFunctionProps const &props = BC.DM.GetDxilFunctionProps(function); + + switch (props.Node.LaunchType) { + case DXIL::NodeLaunchType::Broadcasting: + *OSOverride << "NodeInvocationSelection:DispatchThreadId\n"; + return addThreadIdComparisonProlog(BC, DXIL::OpCode::ThreadId); + + case DXIL::NodeLaunchType::Coalescing: + // The requested coordinates only mean something as a position within the + // group. A coordinate outside the declared group size matches no + // invocation, and a criterion that nothing satisfies leaves the debugger + // with no trace at all. Select every invocation in that case instead. + if (m_Parameters.ComputeShader.ThreadIdX < props.numThreads[0] && + m_Parameters.ComputeShader.ThreadIdY < props.numThreads[1] && + m_Parameters.ComputeShader.ThreadIdZ < props.numThreads[2]) { + *OSOverride << "NodeInvocationSelection:GroupThreadId\n"; + return addThreadIdComparisonProlog(BC, DXIL::OpCode::ThreadIdInGroup); + } + *OSOverride << "NodeInvocationSelection:None\n"; + return BC.HlslOP->GetI1Const(1); + + case DXIL::NodeLaunchType::Thread: + default: + *OSOverride << "NodeInvocationSelection:None\n"; + return BC.HlslOP->GetI1Const(1); + } +} + Value *DxilDebugInstrumentation::addRaygenShaderProlog(BuilderContext &BC) { auto DispatchRaysIndexOpFunc = BC.HlslOP->GetOpFunc( DXIL::OpCode::DispatchRaysIndex, Type::getInt32Ty(BC.Ctx)); @@ -819,10 +881,16 @@ void DxilDebugInstrumentation::addInvocationSelectionProlog( case DXIL::ShaderKind::Intersection: case DXIL::ShaderKind::AnyHit: case DXIL::ShaderKind::Miss: + case DXIL::ShaderKind::Callable: + // A callable shader has neither a ray nor a thread of its own, but it runs + // as part of the dispatch that calls it, so DispatchRaysIndex names the + // invocation that leads to it. PIX selects a raygen invocation on that same + // value, so a callable that the selected raygen thread invokes is selected + // here too. ParameterTestResult = addRaygenShaderProlog(BC); break; case DXIL::ShaderKind::Node: - ParameterTestResult = BC.HlslOP->GetI1Const(1); + ParameterTestResult = addNodeShaderProlog(BC); break; case DXIL::ShaderKind::Compute: case DXIL::ShaderKind::Amplification: diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugCallableShader.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugCallableShader.hlsl new file mode 100644 index 0000000000..7aa4449307 --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugCallableShader.hlsl @@ -0,0 +1,38 @@ +// RUN: %dxc -T lib_6_3 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=1,parameter1=2,parameter2=3 -hlsl-dxilemit | %FileCheck %s + +// A callable shader is a valid CallShader() target, so PIX must be able to step +// into it. The pass numbers its instructions and advertises them to PIX, so it +// must instrument it as well. +// +// A callable shader has no ray and no thread of its own, but +// dx.op.dispatchRaysIndex is legal in it, and reports the index of the ray +// generation invocation that is responsible for the call. PIX selects a raygen, +// any-hit, closest-hit or miss invocation on that same identity, so a callable +// shader uses it too. + +// The callable shader is the only shader in the module, so a Block# line means +// that it is instrumented. +// CHECK: Block# + +// CHECK: %RayX = call i32 @dx.op.dispatchRaysIndex.i32(i32 145, i8 0) +// CHECK: %RayY = call i32 @dx.op.dispatchRaysIndex.i32(i32 145, i8 1) +// CHECK: %RayZ = call i32 @dx.op.dispatchRaysIndex.i32(i32 145, i8 2) +// CHECK: %CompareToThreadIdX = icmp eq i32 %RayX, 1 +// CHECK: %CompareToThreadIdY = icmp eq i32 %RayY, 2 +// CHECK: %CompareToThreadIdZ = icmp eq i32 %RayZ, 3 +// CHECK: %CompareAll = and i1 %CompareXAndY, %CompareToThreadIdZ +// CHECK: br i1 %CompareAll, label %PIXInterestingBlock, label %PIXNonInterestingBlock + +RWStructuredBuffer Output : register(u0); + +struct CallableParameters +{ + float value; +}; + +[shader("callable")] +void MyCallable(inout CallableParameters parameters) +{ + parameters.value = parameters.value * 2.f; + Output[0] = parameters.value; +} diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugNodeBroadcasting.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugNodeBroadcasting.hlsl new file mode 100644 index 0000000000..d48b6d2b0a --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugNodeBroadcasting.hlsl @@ -0,0 +1,35 @@ +// RUN: %dxc -T lib_6_8 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=10,parameter1=20,parameter2=30 -hlsl-dxilemit | %FileCheck %s + +// A broadcasting node runs over a grid, so SV_DispatchThreadID names an +// invocation exactly as it does for a compute shader, and the debugger selects +// the one that the user asks for. A constant-true criterion instead makes every +// invocation believe that it is the selected one, and the debugger shows +// whichever one reaches the UAV first. + +// CHECK: NodeInvocationSelection:DispatchThreadId + +// CHECK: %ThreadIdX = call i32 @dx.op.threadId.i32(i32 93, i32 0) +// CHECK: %ThreadIdY = call i32 @dx.op.threadId.i32(i32 93, i32 1) +// CHECK: %ThreadIdZ = call i32 @dx.op.threadId.i32(i32 93, i32 2) +// CHECK: %CompareToThreadIdX = icmp eq i32 %ThreadIdX, 10 +// CHECK: %CompareToThreadIdY = icmp eq i32 %ThreadIdY, 20 +// CHECK: %CompareToThreadIdZ = icmp eq i32 %ThreadIdZ, 30 +// CHECK: %CompareXAndY = and i1 %CompareToThreadIdX, %CompareToThreadIdY +// CHECK: %CompareAll = and i1 %CompareXAndY, %CompareToThreadIdZ +// CHECK: br i1 %CompareAll, label %PIXInterestingBlock, label %PIXNonInterestingBlock + +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("broadcasting")] +[NodeDispatchGrid(2, 1, 1)] +[NumThreads(4, 2, 1)] +void BroadcastingNode(DispatchNodeInputRecord input) +{ + Output[0] = input.Get().value; +} diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugNodeCoalescing.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugNodeCoalescing.hlsl new file mode 100644 index 0000000000..bea8161684 --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugNodeCoalescing.hlsl @@ -0,0 +1,36 @@ +// RUN: %dxc -T lib_6_8 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=3,parameter1=1,parameter2=0 -hlsl-dxilemit | %FileCheck %s + +// A coalescing node has no dispatch grid, so dx.op.threadId is not legal for it. +// ValidateDxilOperationCallInProfile in DxilValidation.cpp permits ThreadId and +// GroupId for a broadcasting launch only. The thread group ID is legal, so the +// debugger discriminates invocations within a group by SV_GroupThreadID. + +// CHECK: NodeInvocationSelection:GroupThreadId + +// CHECK: %ThreadIdX = call i32 @dx.op.threadIdInGroup.i32(i32 95, i32 0) +// CHECK: %ThreadIdY = call i32 @dx.op.threadIdInGroup.i32(i32 95, i32 1) +// CHECK: %ThreadIdZ = call i32 @dx.op.threadIdInGroup.i32(i32 95, i32 2) +// CHECK: %CompareToThreadIdX = icmp eq i32 %ThreadIdX, 3 +// CHECK: %CompareToThreadIdY = icmp eq i32 %ThreadIdY, 1 +// CHECK: %CompareToThreadIdZ = icmp eq i32 %ThreadIdZ, 0 +// CHECK: %CompareAll = and i1 %CompareXAndY, %CompareToThreadIdZ +// CHECK: br i1 %CompareAll, label %PIXInterestingBlock, label %PIXNonInterestingBlock + +// The requested thread must lie inside the declared thread group, or the pass +// discriminates nothing. The parameters above lie inside [NumThreads(4, 2, 1)]. +// CHECK-NOT: NodeInvocationSelection:None + +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("coalescing")] +[NumThreads(4, 2, 1)] +void CoalescingNode(GroupNodeInputRecords input) +{ + Output[0] = input.Get(0).value; +} diff --git a/tools/clang/test/HLSLFileCheck/pix/DebugNodeThreadLaunch.hlsl b/tools/clang/test/HLSLFileCheck/pix/DebugNodeThreadLaunch.hlsl new file mode 100644 index 0000000000..e5144624bb --- /dev/null +++ b/tools/clang/test/HLSLFileCheck/pix/DebugNodeThreadLaunch.hlsl @@ -0,0 +1,30 @@ +// RUN: %dxc -T lib_6_8 %s | %opt -S -dxil-annotate-with-virtual-regs -hlsl-dxil-debug-instrumentation,parameter0=1,parameter1=2,parameter2=3 -hlsl-dxilemit | %FileCheck %s + +// A thread launch node has neither a dispatch grid nor a thread group, so none +// of the thread-identity intrinsics are legal for it. +// ValidateDxilOperationCallInProfile in DxilValidation.cpp rejects ThreadId, +// GroupId, ThreadIdInGroup and FlattenedThreadIdInGroup for a thread launch +// node. Nothing in the shader tells one invocation from another, so the pass +// keeps the select-everything criterion. This test pins that decision, so that +// a change cannot emit an intrinsic that the validator rejects. + +// CHECK: NodeInvocationSelection:None + +// CHECK-NOT: @dx.op.threadId.i32 +// CHECK-NOT: @dx.op.threadIdInGroup.i32 +// CHECK-NOT: @dx.op.flattenedThreadIdInGroup.i32 +// CHECK: br i1 true, label %PIXInterestingBlock, label %PIXNonInterestingBlock + +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("thread")] +void ThreadLaunchNode(ThreadNodeInputRecord input) +{ + Output[0] = input.Get().value; +} diff --git a/tools/clang/unittests/HLSL/PixTest.cpp b/tools/clang/unittests/HLSL/PixTest.cpp index f232a849d8..2eb11be08b 100644 --- a/tools/clang/unittests/HLSL/PixTest.cpp +++ b/tools/clang/unittests/HLSL/PixTest.cpp @@ -232,6 +232,12 @@ class PixTest : public ::testing::Test { TEST_METHOD(HelperInlining_NonUniformResourceIndexHelperValidates) TEST_METHOD(HelperInlining_DebugBreakHelperValidates) + TEST_METHOD(InvocationSelection_CallableShaderValidates) + TEST_METHOD(InvocationSelection_NodeBroadcastingValidates) + TEST_METHOD(InvocationSelection_NodeCoalescingValidates) + TEST_METHOD(InvocationSelection_NodeCoalescingOutOfGroupSelectsAll) + TEST_METHOD(InvocationSelection_NodeThreadLaunchValidates) + // Control tests for the PIX pass validation harness below // (validateInstrumentedModule / verifyInstrumentedModuleIsValid). TEST_METHOD(Validation_ControlValidModulePasses) @@ -356,6 +362,37 @@ class PixTest : public ::testing::Test { std::move(pOptimizedModule), {}, Tokenize(outputText.c_str(), "\n")}; } + // Runs the debug pass with an explicit invocation to select, so a test drives + // the selection prolog the way PIX does. + PassOutput runDebugPassWithParameters(IDxcBlob *dxil, unsigned parameter0, + unsigned parameter1, + unsigned parameter2) { + CComPtr pOptimizer; + VERIFY_SUCCEEDED( + m_dllSupport.CreateInstance(CLSID_DxcOptimizer, &pOptimizer)); + std::vector Options; + Options.push_back(L"-opt-mod-passes"); + Options.push_back(L"-dxil-dbg-value-to-dbg-declare"); + Options.push_back(L"-dxil-annotate-with-virtual-regs"); + std::wstring debugArg = L"-hlsl-dxil-debug-instrumentation,parameter0=" + + std::to_wstring(parameter0) + L",parameter1=" + + std::to_wstring(parameter1) + L",parameter2=" + + std::to_wstring(parameter2); + Options.push_back(debugArg.c_str()); + Options.push_back(L"-viewid-state"); + Options.push_back(L"-hlsl-dxilemit"); + + CComPtr pOptimizedModule; + CComPtr pText; + VERIFY_SUCCEEDED(pOptimizer->RunOptimizer( + dxil, Options.data(), Options.size(), &pOptimizedModule, &pText)); + + std::string outputText = BlobToUtf8(pText); + + return { + std::move(pOptimizedModule), {}, Tokenize(outputText.c_str(), "\n")}; + } + PassOutput RunDebugBreakPass(IDxcBlob *dxil) { CComPtr pOptimizer; VERIFY_SUCCEEDED( @@ -6657,3 +6694,192 @@ TEST_F(PixTest, HelperInlining_InlinedHelperValidates) { output.blob, "debug instrumentation of a shader whose [noinline] helper " "is inlined away"); } + +// A callable shader has no ray and no thread of its own, but +// dx.op.dispatchRaysIndex is legal in it and reports the ray generation index +// that is responsible for the call. PIX selects a raygen invocation on that +// same identity, so a callable uses it too. +TEST_F(PixTest, InvocationSelection_CallableShaderValidates) { + const char *source = R"x( +RWStructuredBuffer Output : register(u0); + +struct CallableParameters +{ + float value; +}; + +[shader("callable")] +void MyCallable(inout CallableParameters parameters) +{ + parameters.value = parameters.value * 2.f; + Output[0] = parameters.value; +})x"; + + auto compiled = Compile(m_dllSupport, source, L"lib_6_3", {L"-Od"}); + auto output = RunDebugPass(compiled); + std::string disassembly = Disassemble(output.blob); + + // The callable is the only shader in the module, so a block report means that + // it is instrumented. + VERIFY_IS_TRUE(AnyLineContains(output.lines, "Block#")); + VERIFY_ARE_EQUAL(3u, + CountCallsTo(disassembly, "@dx.op.dispatchRaysIndex.i32")); + VERIFY_ARE_EQUAL(1u, CountOccurrences(disassembly, "PIXInterestingBlock:")); + + verifyInstrumentedModuleIsValid(output.blob, + "debug instrumentation of a callable shader"); +} + +// A broadcasting node runs over a grid, so SV_DispatchThreadID names an +// invocation exactly as it does for a compute shader. +TEST_F(PixTest, InvocationSelection_NodeBroadcastingValidates) { + const char *source = R"x( +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("broadcasting")] +[NodeDispatchGrid(2, 1, 1)] +[NumThreads(4, 2, 1)] +void BroadcastingNode(DispatchNodeInputRecord input) +{ + Output[0] = input.Get().value; +})x"; + + if (m_ver.SkipDxilVersion(1, 8)) { + return; + } + + auto compiled = Compile(m_dllSupport, source, L"lib_6_8", {L"-Od"}); + auto output = RunDebugPass(compiled); + std::string disassembly = Disassemble(output.blob); + + VERIFY_IS_TRUE(AnyLineContains(output.lines, + "NodeInvocationSelection:DispatchThreadId")); + VERIFY_ARE_EQUAL(3u, CountCallsTo(disassembly, "@dx.op.threadId.i32")); + VERIFY_ARE_EQUAL(0u, CountCallsTo(disassembly, "@dx.op.threadIdInGroup.i32")); + + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of a broadcasting node shader"); +} + +// A coalescing node has no dispatch grid, so the validator rejects ThreadId for +// it. The thread group ID is legal, and discriminates invocations within one +// group. +TEST_F(PixTest, InvocationSelection_NodeCoalescingValidates) { + const char *source = R"x( +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("coalescing")] +[NumThreads(4, 2, 1)] +void CoalescingNode(GroupNodeInputRecords input) +{ + Output[0] = input.Get(0).value; +})x"; + + if (m_ver.SkipDxilVersion(1, 8)) { + return; + } + + auto compiled = Compile(m_dllSupport, source, L"lib_6_8", {L"-Od"}); + // The requested thread lies inside [NumThreads(4, 2, 1)]. + auto output = runDebugPassWithParameters(compiled, 3, 1, 0); + std::string disassembly = Disassemble(output.blob); + + VERIFY_IS_TRUE( + AnyLineContains(output.lines, "NodeInvocationSelection:GroupThreadId")); + VERIFY_ARE_EQUAL(3u, CountCallsTo(disassembly, "@dx.op.threadIdInGroup.i32")); + VERIFY_ARE_EQUAL(0u, CountCallsTo(disassembly, "@dx.op.threadId.i32")); + + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of a coalescing node shader"); +} + +// A coordinate outside the declared group size matches no invocation, and a +// criterion that nothing satisfies leaves the debugger with no trace at all. +// Every invocation stays of interest instead. +TEST_F(PixTest, InvocationSelection_NodeCoalescingOutOfGroupSelectsAll) { + const char *source = R"x( +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("coalescing")] +[NumThreads(4, 2, 1)] +void CoalescingNode(GroupNodeInputRecords input) +{ + Output[0] = input.Get(0).value; +})x"; + + if (m_ver.SkipDxilVersion(1, 8)) { + return; + } + + auto compiled = Compile(m_dllSupport, source, L"lib_6_8", {L"-Od"}); + // Y is 5, which lies outside [NumThreads(4, 2, 1)]. + auto output = runDebugPassWithParameters(compiled, 1, 5, 0); + std::string disassembly = Disassemble(output.blob); + + VERIFY_IS_TRUE(AnyLineContains(output.lines, "NodeInvocationSelection:None")); + VERIFY_ARE_EQUAL(0u, CountCallsTo(disassembly, "@dx.op.threadIdInGroup.i32")); + VERIFY_IS_TRUE(disassembly.find("br i1 true, label %PIXInterestingBlock") != + std::string::npos); + + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of a coalescing node shader whose " + "requested thread lies outside the group"); +} + +// A thread launch node has no thread-identity system value at all, so nothing +// distinguishes one invocation from another and every invocation stays of +// interest. This pins that decision, so that no intrinsic the validator rejects +// is emitted. +TEST_F(PixTest, InvocationSelection_NodeThreadLaunchValidates) { + const char *source = R"x( +RWStructuredBuffer Output : register(u0); + +struct Record +{ + uint value; +}; + +[Shader("node")] +[NodeLaunch("thread")] +void ThreadLaunchNode(ThreadNodeInputRecord input) +{ + Output[0] = input.Get().value; +})x"; + + if (m_ver.SkipDxilVersion(1, 8)) { + return; + } + + auto compiled = Compile(m_dllSupport, source, L"lib_6_8", {L"-Od"}); + auto output = RunDebugPass(compiled); + std::string disassembly = Disassemble(output.blob); + + VERIFY_IS_TRUE(AnyLineContains(output.lines, "NodeInvocationSelection:None")); + VERIFY_ARE_EQUAL(0u, CountCallsTo(disassembly, "@dx.op.threadId.i32")); + VERIFY_ARE_EQUAL(0u, CountCallsTo(disassembly, "@dx.op.threadIdInGroup.i32")); + VERIFY_ARE_EQUAL( + 0u, CountCallsTo(disassembly, "@dx.op.flattenedThreadIdInGroup.i32")); + VERIFY_IS_TRUE(disassembly.find("br i1 true, label %PIXInterestingBlock") != + std::string::npos); + + verifyInstrumentedModuleIsValid( + output.blob, "debug instrumentation of a thread launch node shader"); +}