diff --git a/renderdoc/driver/d3d12/d3d12_shaderdebug.cpp b/renderdoc/driver/d3d12/d3d12_shaderdebug.cpp index d5591e0ff..da3f206a0 100644 --- a/renderdoc/driver/d3d12/d3d12_shaderdebug.cpp +++ b/renderdoc/driver/d3d12/d3d12_shaderdebug.cpp @@ -2005,11 +2005,15 @@ ShaderDebugTrace *D3D12Replay::DebugVertex(uint32_t eventId, uint32_t vertid, ui else { DXILDebug::Debugger *debugger = new DXILDebug::Debugger(); - ret = debugger->BeginDebug(eventId, dxbc, refl, 0); + ret = debugger->BeginDebug(eventId, dxbc, refl, 0, 1); DXILDebug::GlobalState &globalState = debugger->GetGlobalState(); DXILDebug::ThreadState &activeState = debugger->GetActiveLane(); rdcarray &inputs = activeState.m_Input.members; + rdcarray workgroupProperties; + workgroupProperties.resize(1); + + workgroupProperties[0][DXILDebug::ThreadProperty::Active] = 1; // Fetch constant buffer data from root signature DXILDebug::FetchConstantBufferData(m_pDevice, dxbc->GetDXILByteCode(), rs.graphics, refl, @@ -2258,6 +2262,9 @@ ShaderDebugTrace *D3D12Replay::DebugVertex(uint32_t eventId, uint32_t vertid, ui default: RDCERR("Unhandled system value semantic on VS input"); break; } } + + debugger->InitialiseWorkgroup(workgroupProperties); + ret->inputs = {activeState.m_Input}; ret->constantBlocks = globalState.constantBlocks; delete[] instData; @@ -2853,9 +2860,11 @@ ShaderDebugTrace *D3D12Replay::DebugPixel(uint32_t eventId, uint32_t x, uint32_t else { DXILDebug::Debugger *debugger = new DXILDebug::Debugger(); - ret = debugger->BeginDebug(eventId, dxbc, refl, hit->quadLaneIndex); + ret = debugger->BeginDebug(eventId, dxbc, refl, hit->quadLaneIndex, 4); DXILDebug::GlobalState &globalState = debugger->GetGlobalState(); + rdcarray workgroupProperties; + workgroupProperties.resize(4); const rdcarray &dxilInputs = debugger->GetDXILEntryPointInputs(); @@ -2870,8 +2879,12 @@ ShaderDebugTrace *D3D12Replay::DebugPixel(uint32_t eventId, uint32_t x, uint32_t DXILDebug::ThreadState &state = debugger->GetWorkgroup(q); rdcarray &ins = state.m_Input.members; - if(q != hit->quadLaneIndex) - state.InitialiseHelper(debugger->GetActiveLane()); + RDCASSERT(q == lane->quadLane, q, lane->quadLane); + + workgroupProperties[q][DXILDebug::ThreadProperty::Active] = 1; + workgroupProperties[q][DXILDebug::ThreadProperty::Helper] = q != hit->quadLaneIndex; + workgroupProperties[q][DXILDebug::ThreadProperty::QuadLane] = lane->quadLane; + workgroupProperties[q][DXILDebug::ThreadProperty::QuadId] = lane->quadId; data += sizeof(DXDebug::LaneData); @@ -2976,6 +2989,8 @@ ShaderDebugTrace *D3D12Replay::DebugPixel(uint32_t eventId, uint32_t x, uint32_t #endif } + debugger->InitialiseWorkgroup(workgroupProperties); + ret->inputs = {debugger->GetActiveLane().m_Input}; ret->constantBlocks = globalState.constantBlocks; } @@ -3121,10 +3136,12 @@ ShaderDebugTrace *D3D12Replay::DebugThread(uint32_t eventId, m_pDevice->ReplayLog(0, eventId, eReplay_WithoutDraw); DXILDebug::Debugger *debugger = new DXILDebug::Debugger(); - ret = debugger->BeginDebug(eventId, dxbc, refl, 0); + ret = debugger->BeginDebug(eventId, dxbc, refl, 0, 1); DXILDebug::GlobalState &globalState = debugger->GetGlobalState(); std::map &builtins = globalState.builtinInputs; + rdcarray workgroupProperties; + workgroupProperties.resize(1); uint32_t threadDim[3] = { refl.dispatchThreadsDimension[0], @@ -3132,6 +3149,8 @@ ShaderDebugTrace *D3D12Replay::DebugThread(uint32_t eventId, refl.dispatchThreadsDimension[2], }; + workgroupProperties[0][DXILDebug::ThreadProperty::Active] = 1; + // SV_DispatchThreadID builtins[ShaderBuiltin::DispatchThreadIndex] = ShaderVariable( rdcstr(), groupid[0] * threadDim[0] + threadid[0], groupid[1] * threadDim[1] + threadid[1], @@ -3154,6 +3173,9 @@ ShaderDebugTrace *D3D12Replay::DebugThread(uint32_t eventId, // Fetch constant buffer data from root signature DXILDebug::FetchConstantBufferData(m_pDevice, dxbc->GetDXILByteCode(), rs.compute, refl, globalState, ret->sourceVars); + + debugger->InitialiseWorkgroup(workgroupProperties); + // ret->inputs = state.inputs; ret->constantBlocks = globalState.constantBlocks; } diff --git a/renderdoc/driver/shaders/dxbc/dx_debug.cpp b/renderdoc/driver/shaders/dxbc/dx_debug.cpp index aaca5718e..7bac43944 100644 --- a/renderdoc/driver/shaders/dxbc/dx_debug.cpp +++ b/renderdoc/driver/shaders/dxbc/dx_debug.cpp @@ -492,7 +492,7 @@ void ExtractInputsPS(PSInput IN, // quadId is a single value that's unique for this quad and uniform across the quad. Degenerate // for the simple quad case - uint quadId = quadSwizzleHelper(quadLaneIndex, quadLaneIndex, 0u); + uint quadId = 1000+quadSwizzleHelper(quadLaneIndex, quadLaneIndex, 0u); HitBuffer[idx].quad[0].quadId = quadId; HitBuffer[idx].quad[1].quadId = quadId; HitBuffer[idx].quad[2].quadId = quadId; diff --git a/renderdoc/driver/shaders/dxil/dxil_debug.cpp b/renderdoc/driver/shaders/dxil/dxil_debug.cpp index 67316246b..bd0fcd48c 100644 --- a/renderdoc/driver/shaders/dxil/dxil_debug.cpp +++ b/renderdoc/driver/shaders/dxil/dxil_debug.cpp @@ -1493,24 +1493,13 @@ void MemoryTracking::AllocateMemoryForType(const DXIL::Type *type, Id allocId, b m_Pointers[allocId] = {allocId, backingMem, byteSize}; } -ThreadState::ThreadState(uint32_t workgroupIndex, Debugger &debugger, - const GlobalState &globalState, uint32_t maxSSAId) +ThreadState::ThreadState(Debugger &debugger, const GlobalState &globalState, uint32_t maxSSAId) : m_Debugger(debugger), m_GlobalState(globalState), m_Program(debugger.GetProgram()), m_MaxSSAId(maxSSAId) { - m_WorkgroupIndex = workgroupIndex; - m_FunctionInfo = NULL; - m_FunctionInstructionIdx = 0; - m_ActiveGlobalInstructionIdx = 0; - m_Killed = false; - m_Ended = false; - m_Callstack.clear(); m_ShaderType = m_Program.GetShaderType(); - m_Semantics.coverage = ~0U; - m_Semantics.isFrontFace = false; - m_Semantics.primID = ~0U; m_Assigned.resize(maxSSAId); m_Live.resize(maxSSAId); } @@ -1524,19 +1513,9 @@ ThreadState::~ThreadState() } } -void ThreadState::InitialiseHelper(const ThreadState &activeState) -{ - m_Input = activeState.m_Input; - m_Semantics = activeState.m_Semantics; - m_Variables = activeState.m_Variables; - m_Assigned = activeState.m_Assigned; - m_Live = activeState.m_Live; - m_IsGlobal = activeState.m_IsGlobal; -} - bool ThreadState::Finished() const { - return m_Killed || m_Ended || m_Callstack.empty(); + return m_Dead || m_Ended || m_Callstack.empty(); } bool ThreadState::InUniformBlock() const @@ -1676,7 +1655,7 @@ bool IsNopInstruction(const Instruction &inst) } bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, - const rdcarray &workgroups) + const rdcarray &workgroup) { m_CurrentInstruction = m_FunctionInfo->function->instructions[m_FunctionInstructionIdx]; const Instruction &inst = *m_CurrentInstruction; @@ -1891,7 +1870,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, if(!resRefInfo.Valid()) break; - PerformGPUResourceOp(workgroups, opCode, dxOpCode, resRefInfo, apiWrapper, inst, result); + PerformGPUResourceOp(workgroup, opCode, dxOpCode, resRefInfo, apiWrapper, inst, result); eventFlags |= ShaderEvents::SampleLoadGather; break; } @@ -1918,8 +1897,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, // SRV TextureLoad is done on the GPU if((dxOpCode == DXOp::TextureLoad) && (resClass == ResourceClass::SRV)) { - PerformGPUResourceOp(workgroups, opCode, dxOpCode, resRefInfo, apiWrapper, inst, - result); + PerformGPUResourceOp(workgroup, opCode, dxOpCode, resRefInfo, apiWrapper, inst, result); eventFlags |= ShaderEvents::SampleLoadGather; break; } @@ -2755,21 +2733,21 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, case DXOp::DerivFineX: case DXOp::DerivFineY: { - if(m_ShaderType != DXBC::ShaderType::Pixel || workgroups.size() != 4) + if(m_ShaderType != DXBC::ShaderType::Pixel || workgroup.size() != 4) { RDCERR("Undefined results using derivative instruction outside of a pixel shader."); } else { - RDCASSERT(!ThreadsAreDiverged(workgroups)); + RDCASSERT(!QuadIsDiverged(workgroup, m_QuadNeighbours)); if(dxOpCode == DXOp::DerivCoarseX) - result.value = DDX(false, opCode, dxOpCode, workgroups, inst.args[1]); + result.value = DDX(false, opCode, dxOpCode, workgroup, inst.args[1]); else if(dxOpCode == DXOp::DerivCoarseY) - result.value = DDY(false, opCode, dxOpCode, workgroups, inst.args[1]); + result.value = DDY(false, opCode, dxOpCode, workgroup, inst.args[1]); else if(dxOpCode == DXOp::DerivFineX) - result.value = DDX(true, opCode, dxOpCode, workgroups, inst.args[1]); + result.value = DDX(true, opCode, dxOpCode, workgroup, inst.args[1]); else if(dxOpCode == DXOp::DerivFineY) - result.value = DDY(true, opCode, dxOpCode, workgroups, inst.args[1]); + result.value = DDY(true, opCode, dxOpCode, workgroup, inst.args[1]); } break; } @@ -2913,7 +2891,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, BarrierMode barrierMode = (BarrierMode)arg.value.u32v[0]; // For thread barriers the threads must be converged if(barrierMode & BarrierMode::SyncThreadGroup) - RDCASSERT(!ThreadsAreDiverged(workgroups)); + RDCASSERT(!WorkgroupIsDiverged(workgroup)); break; } case DXOp::Discard: @@ -2922,7 +2900,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, RDCASSERT(GetShaderVariable(inst.args[1], opCode, dxOpCode, cond)); if(cond.value.u32v[0] != 0) { - m_Killed = true; + m_Dead = true; return true; } break; @@ -3295,8 +3273,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, } case DXOp::IsHelperLane: { - // Helper lanes don't have state - result.value.u32v[0] = m_State ? 0 : 1; + result.value.u32v[0] = m_Helper ? 0 : 1; break; } case DXOp::UAddc: @@ -3573,7 +3550,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, case DXOp::QuadReadLaneAt: case DXOp::QuadOp: { - RDCASSERT(!ThreadsAreDiverged(workgroups)); + RDCASSERT(!QuadIsDiverged(workgroup, m_QuadNeighbours)); // QuadOp(value,op) // QuadReadLaneAt(value,quadLane) ShaderVariable b; @@ -3629,10 +3606,10 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, { RDCERR("Unhandled dxOpCode %s", ToStr(dxOpCode).c_str()); } - if(lane < workgroups.size()) + if(lane < workgroup.size()) { ShaderVariable var; - RDCASSERT(workgroups[lane].GetShaderVariable(inst.args[1], opCode, dxOpCode, var)); + RDCASSERT(workgroup[lane].GetShaderVariable(inst.args[1], opCode, dxOpCode, var)); result.value = var.value; } else @@ -3992,7 +3969,7 @@ bool ThreadState::ExecuteInstruction(DebugAPIWrapper *apiWrapper, case Operation::NoOp: RDCERR("NoOp instructions should not be executed"); return false; case Operation::Unreachable: { - m_Killed = true; + m_Dead = true; RDCERR("Operation::Unreachable reached, terminating debugging!"); return true; } @@ -5361,7 +5338,7 @@ void ThreadState::StepOverNopInstructions() } void ThreadState::StepNext(ShaderDebugState *state, DebugAPIWrapper *apiWrapper, - const rdcarray &workgroups) + const rdcarray &workgroup) { m_State = state; @@ -5399,7 +5376,7 @@ void ThreadState::StepNext(ShaderDebugState *state, DebugAPIWrapper *apiWrapper, } } } - ExecuteInstruction(apiWrapper, workgroups); + ExecuteInstruction(apiWrapper, workgroup); m_State = NULL; } @@ -5704,7 +5681,7 @@ void ThreadState::UpdateMemoryVariableFromBackingMemory(Id memoryId, const void } } -void ThreadState::PerformGPUResourceOp(const rdcarray &workgroups, Operation opCode, +void ThreadState::PerformGPUResourceOp(const rdcarray &workgroup, Operation opCode, DXOp dxOpCode, const ResourceReferenceInfo &resRefInfo, DebugAPIWrapper *apiWrapper, const DXIL::Instruction &inst, ShaderVariable &result) @@ -5935,22 +5912,22 @@ void ThreadState::PerformGPUResourceOp(const rdcarray &workgroups, // Sample, SampleBias, CalculateLOD need DDX, DDY if((dxOpCode == DXOp::Sample) || (dxOpCode == DXOp::SampleBias) || (dxOpCode == DXOp::CalculateLOD)) { - if(m_ShaderType != DXBC::ShaderType::Pixel || workgroups.size() != 4) + if(m_ShaderType != DXBC::ShaderType::Pixel || m_QuadNeighbours.contains(~0U)) { RDCERR("Undefined results using derivative instruction outside of a pixel shader."); } else { - RDCASSERT(!ThreadsAreDiverged(workgroups)); + RDCASSERT(!QuadIsDiverged(workgroup, m_QuadNeighbours)); // texture samples use coarse derivatives ShaderValue delta; for(uint32_t i = 0; i < 4; i++) { if(uvDDXY[i]) { - delta = DDX(false, opCode, dxOpCode, workgroups, inst.args[3 + i]); + delta = DDX(false, opCode, dxOpCode, workgroup, inst.args[3 + i]); ddx.value.f32v[i] = delta.f32v[0]; - delta = DDY(false, opCode, dxOpCode, workgroups, inst.args[3 + i]); + delta = DDY(false, opCode, dxOpCode, workgroup, inst.args[3 + i]); ddy.value.f32v[i] = delta.f32v[0]; } } @@ -6107,12 +6084,25 @@ void ThreadState::Sub(const ShaderVariable &a, const ShaderVariable &b, ShaderVa } ShaderValue ThreadState::DDX(bool fine, Operation opCode, DXOp dxOpCode, - const rdcarray &quad, const DXIL::Value *dxilValue) const + const rdcarray &workgroup, const DXIL::Value *dxilValue) const { - RDCASSERT(!ThreadsAreDiverged(quad)); + ShaderValue ret = {}; + + if(m_QuadNeighbours[0] == ~0U || m_QuadNeighbours[1] == ~0U || m_QuadNeighbours[2] == ~0U || + m_QuadNeighbours[3] == ~0U) + { + RDCERR("Derivative calculation within non-quad"); + return ret; + } + + RDCASSERT(m_QuadNeighbours[0] < workgroup.size(), m_QuadNeighbours[0], workgroup.size()); + RDCASSERT(m_QuadNeighbours[1] < workgroup.size(), m_QuadNeighbours[1], workgroup.size()); + RDCASSERT(m_QuadNeighbours[2] < workgroup.size(), m_QuadNeighbours[2], workgroup.size()); + RDCASSERT(m_QuadNeighbours[3] < workgroup.size(), m_QuadNeighbours[3], workgroup.size()); + RDCASSERT(!QuadIsDiverged(workgroup, m_QuadNeighbours)); uint32_t index = ~0U; - int quadIndex = m_WorkgroupIndex; + int quadIndex = m_QuadLaneIndex; if(!fine) { @@ -6129,21 +6119,34 @@ ShaderValue ThreadState::DDX(bool fine, Operation opCode, DXOp dxOpCode, index = quadIndex - 1; } - ShaderValue ret; ShaderVariable a; ShaderVariable b; - RDCASSERT(quad[index + 1].GetShaderVariable(dxilValue, opCode, dxOpCode, a)); - RDCASSERT(quad[index].GetShaderVariable(dxilValue, opCode, dxOpCode, b)); + RDCASSERT(workgroup[m_QuadNeighbours[index + 1]].GetShaderVariable(dxilValue, opCode, dxOpCode, a)); + RDCASSERT(workgroup[m_QuadNeighbours[index]].GetShaderVariable(dxilValue, opCode, dxOpCode, b)); Sub(a, b, ret); return ret; } ShaderValue ThreadState::DDY(bool fine, Operation opCode, DXOp dxOpCode, - const rdcarray &quad, const DXIL::Value *dxilValue) const + const rdcarray &workgroup, const DXIL::Value *dxilValue) const { - RDCASSERT(!ThreadsAreDiverged(quad)); + ShaderValue ret = {}; + + if(m_QuadNeighbours[0] == ~0U || m_QuadNeighbours[1] == ~0U || m_QuadNeighbours[2] == ~0U || + m_QuadNeighbours[3] == ~0U) + { + RDCERR("Derivative calculation within non-quad"); + return ret; + } + + RDCASSERT(m_QuadNeighbours[0] < workgroup.size(), m_QuadNeighbours[0], workgroup.size()); + RDCASSERT(m_QuadNeighbours[1] < workgroup.size(), m_QuadNeighbours[1], workgroup.size()); + RDCASSERT(m_QuadNeighbours[2] < workgroup.size(), m_QuadNeighbours[2], workgroup.size()); + RDCASSERT(m_QuadNeighbours[3] < workgroup.size(), m_QuadNeighbours[3], workgroup.size()); + RDCASSERT(!QuadIsDiverged(workgroup, m_QuadNeighbours)); + uint32_t index = ~0U; - int quadIndex = m_WorkgroupIndex; + int quadIndex = m_QuadLaneIndex; if(!fine) { @@ -6160,12 +6163,10 @@ ShaderValue ThreadState::DDY(bool fine, Operation opCode, DXOp dxOpCode, index = quadIndex - 2; } - ShaderValue ret; - memset(&ret, 0, sizeof(ret)); ShaderVariable a; ShaderVariable b; - RDCASSERT(quad[index + 2].GetShaderVariable(dxilValue, opCode, dxOpCode, a)); - RDCASSERT(quad[index].GetShaderVariable(dxilValue, opCode, dxOpCode, b)); + RDCASSERT(workgroup[m_QuadNeighbours[index + 2]].GetShaderVariable(dxilValue, opCode, dxOpCode, a)); + RDCASSERT(workgroup[m_QuadNeighbours[index]].GetShaderVariable(dxilValue, opCode, dxOpCode, b)); Sub(a, b, ret); return ret; } @@ -6179,25 +6180,57 @@ GlobalState::~GlobalState() } } -bool ThreadState::ThreadsAreDiverged(const rdcarray &workgroups) +bool ThreadState::WorkgroupIsDiverged(const rdcarray &workgroup) { uint32_t block0 = ~0U; uint32_t instr0 = ~0U; - for(size_t i = 0; i < workgroups.size(); i++) + for(size_t i = 0; i < workgroup.size(); i++) { - if(workgroups[i].Finished()) + if(workgroup[i].Finished()) continue; if(block0 == ~0U) { - block0 = workgroups[i].m_Block; - instr0 = workgroups[i].m_ActiveGlobalInstructionIdx; + block0 = workgroup[i].m_Block; + instr0 = workgroup[i].m_ActiveGlobalInstructionIdx; continue; } // not in the same basic block - if(workgroups[i].m_Block != block0) + if(workgroup[i].m_Block != block0) return true; // not executing the same instruction - if(workgroups[i].m_ActiveGlobalInstructionIdx != instr0) + if(workgroup[i].m_ActiveGlobalInstructionIdx != instr0) + return true; + } + return false; +} + +bool ThreadState::QuadIsDiverged(const rdcarray &workgroup, + const rdcfixedarray &quadNeighbours) +{ + uint32_t block0 = ~0U; + uint32_t instr0 = ~0U; + for(size_t q = 0; q < quadNeighbours.size(); q++) + { + uint32_t i = quadNeighbours[q]; + if(i == ~0U) + { + RDCERR("Checking quad divergence on non-quad"); + continue; + } + + if(workgroup[i].Finished()) + continue; + if(block0 == ~0U) + { + block0 = workgroup[i].m_Block; + instr0 = workgroup[i].m_ActiveGlobalInstructionIdx; + continue; + } + // not in the same basic block + if(workgroup[i].m_Block != block0) + return true; + // not executing the same instruction + if(workgroup[i].m_ActiveGlobalInstructionIdx != instr0) return true; } return false; @@ -6237,35 +6270,35 @@ rdcstr Debugger::GetResourceReferenceName(const DXIL::Program *program, void Debugger::CalcActiveMask(rdcarray &activeMask) { // one bool per workgroup thread - activeMask.resize(m_Workgroups.size()); + activeMask.resize(m_Workgroup.size()); // mark any threads that have finished as inactive, otherwise they're active - for(size_t i = 0; i < m_Workgroups.size(); i++) - activeMask[i] = !m_Workgroups[i].Finished(); + for(size_t i = 0; i < m_Workgroup.size(); i++) + activeMask[i] = !m_Workgroup[i].Finished(); // only pixel shaders automatically converge workgroups, compute shaders need explicit sync if(m_Stage != ShaderStage::Pixel) return; // Not diverged then all active - if(!ThreadState::ThreadsAreDiverged(m_Workgroups)) + if(!ThreadState::WorkgroupIsDiverged(m_Workgroup)) return; bool anyActive = false; - for(size_t i = 0; i < m_Workgroups.size(); i++) + for(size_t i = 0; i < m_Workgroup.size(); i++) { if(!activeMask[i]) continue; // Run any thread that is not in a uniform block // Stop any thread that is not in a uniform block - activeMask[i] = !m_Workgroups[i].InUniformBlock(); + activeMask[i] = !m_Workgroup[i].InUniformBlock(); anyActive |= activeMask[i]; } if(!anyActive) { RDCERR("No active threads, forcing all unfinished threads to run"); - for(size_t i = 0; i < m_Workgroups.size(); i++) - activeMask[i] = !m_Workgroups[i].Finished(); + for(size_t i = 0; i < m_Workgroup.size(); i++) + activeMask[i] = !m_Workgroup[i].Finished(); } return; } @@ -7450,7 +7483,8 @@ void Debugger::ParseDebugData() } ShaderDebugTrace *Debugger::BeginDebug(uint32_t eventId, const DXBC::DXBCContainer *dxbcContainer, - const ShaderReflection &reflection, uint32_t activeLaneIndex) + const ShaderReflection &reflection, uint32_t activeLaneIndex, + uint32_t workgroupSize) { ShaderStage shaderStage = reflection.stage; @@ -7466,9 +7500,8 @@ ShaderDebugTrace *Debugger::BeginDebug(uint32_t eventId, const DXBC::DXBCContain ShaderDebugTrace *ret = new ShaderDebugTrace; ret->stage = shaderStage; - uint32_t workgroupSize = shaderStage == ShaderStage::Pixel ? 4 : 1; for(uint32_t i = 0; i < workgroupSize; i++) - m_Workgroups.push_back(ThreadState(i, *this, m_GlobalState, nextSSAId)); + m_Workgroup.push_back(ThreadState(*this, m_GlobalState, nextSSAId)); ThreadState &state = GetActiveLane(); @@ -8287,6 +8320,22 @@ ShaderDebugTrace *Debugger::BeginDebug(uint32_t eventId, const DXBC::DXBCContain ret->samplers = m_GlobalState.samplers; ret->debugger = this; + for(uint32_t i = 0; i < workgroupSize; i++) + { + ThreadState &lane = m_Workgroup[i]; + lane.m_WorkgroupIndex = i; + + if(i != m_ActiveLaneIndex) + { + lane.m_Input = state.m_Input; + lane.m_Semantics = state.m_Semantics; + lane.m_Variables = state.m_Variables; + lane.m_Assigned = state.m_Assigned; + lane.m_Live = state.m_Live; + lane.m_IsGlobal = state.m_IsGlobal; + } + } + // Add the output struct to the global state if(countOutputs) m_GlobalState.globals.push_back(state.m_Output); @@ -8294,6 +8343,85 @@ ShaderDebugTrace *Debugger::BeginDebug(uint32_t eventId, const DXBC::DXBCContain return ret; } +void Debugger::InitialiseWorkgroup(const rdcarray &workgroupProperties) +{ + const uint32_t workgroupSize = (uint32_t)m_Workgroup.size(); + + if(workgroupSize == 1) + return; + + if(workgroupSize != workgroupProperties.size()) + { + RDCERR("Workgroup properties has wrong count %zu, expected %u", workgroupProperties.size(), + workgroupSize); + return; + } + + for(uint32_t i = 0; i < workgroupSize; i++) + { + ThreadState &lane = m_Workgroup[i]; + + if(m_Stage == ShaderStage::Pixel) + { + lane.m_Helper = workgroupProperties[i][ThreadProperty::Helper] != 0; + lane.m_QuadLaneIndex = workgroupProperties[i][ThreadProperty::QuadLane]; + lane.m_QuadId = workgroupProperties[i][ThreadProperty::QuadId]; + } + + lane.m_Dead = workgroupProperties[i][ThreadProperty::Active] == 0; + } + + // find quad neighbours + { + rdcarray processedQuads; + for(uint32_t i = 0; i < workgroupSize; i++) + { + uint32_t desiredQuad = m_Workgroup[i].m_QuadId; + + // ignore threads not in any quad + if(desiredQuad == 0) + continue; + + // quads are almost certainly sorted together, so shortcut by checking the last one + if((!processedQuads.empty() && processedQuads.back() == desiredQuad) || + processedQuads.contains(desiredQuad)) + continue; + + processedQuads.push_back(desiredQuad); + + // find the threads + uint32_t threads[4] = { + i, + ~0U, + ~0U, + ~0U, + }; + for(uint32_t j = i + 1, t = 1; j < workgroupSize && t < 4; j++) + { + if(m_Workgroup[j].m_QuadId == desiredQuad) + threads[t++] = j; + } + + // now swizzle the threads to know each other + for(uint32_t src = 0; src < 4; src++) + { + uint32_t lane = m_Workgroup[threads[src]].m_QuadLaneIndex; + + if(lane >= 4) + continue; + + for(uint32_t dst = 0; dst < 4; dst++) + { + if(threads[dst] == ~0U) + continue; + + m_Workgroup[threads[dst]].m_QuadNeighbours[lane] = threads[src]; + } + } + } + } +} + rdcarray Debugger::ContinueDebug(DebugAPIWrapper *apiWrapper) { ThreadState &active = GetActiveLane(); @@ -8305,9 +8433,9 @@ rdcarray Debugger::ContinueDebug(DebugAPIWrapper *apiWrapper) { ShaderDebugState initial; - for(size_t lane = 0; lane < m_Workgroups.size(); lane++) + for(size_t lane = 0; lane < m_Workgroup.size(); lane++) { - ThreadState &thread = m_Workgroups[lane]; + ThreadState &thread = m_Workgroup[lane]; if(lane == m_ActiveLaneIndex) { @@ -8351,11 +8479,11 @@ rdcarray Debugger::ContinueDebug(DebugAPIWrapper *apiWrapper) // step all active members of the workgroup ShaderDebugState state; bool hasDebugState = false; - for(size_t lane = 0; lane < m_Workgroups.size(); lane++) + for(size_t lane = 0; lane < m_Workgroup.size(); lane++) { if(activeMask[lane]) { - ThreadState &thread = m_Workgroups[lane]; + ThreadState &thread = m_Workgroup[lane]; if(thread.Finished()) { if(lane == m_ActiveLaneIndex) @@ -8367,24 +8495,24 @@ rdcarray Debugger::ContinueDebug(DebugAPIWrapper *apiWrapper) { hasDebugState = true; state.stepIndex = m_Steps; - thread.StepNext(&state, apiWrapper, m_Workgroups); + thread.StepNext(&state, apiWrapper, m_Workgroup); m_Steps++; } else { - thread.StepNext(NULL, apiWrapper, m_Workgroups); + thread.StepNext(NULL, apiWrapper, m_Workgroup); } } } - for(size_t lane = 0; lane < m_Workgroups.size(); lane++) + for(size_t lane = 0; lane < m_Workgroup.size(); lane++) { if(activeMask[lane]) - m_Workgroups[lane].StepOverNopInstructions(); + m_Workgroup[lane].StepOverNopInstructions(); } // Update UI state after the execute and step over nops to make sure state.nextInstruction is in sync if(hasDebugState) { - ThreadState &thread = m_Workgroups[m_ActiveLaneIndex]; + ThreadState &thread = m_Workgroup[m_ActiveLaneIndex]; state.nextInstruction = thread.m_ActiveGlobalInstructionIdx; thread.FillCallstack(state); ret.push_back(std::move(state)); diff --git a/renderdoc/driver/shaders/dxil/dxil_debug.h b/renderdoc/driver/shaders/dxil/dxil_debug.h index 25484e8af..4d01e4ff0 100644 --- a/renderdoc/driver/shaders/dxil/dxil_debug.h +++ b/renderdoc/driver/shaders/dxil/dxil_debug.h @@ -214,20 +214,19 @@ struct MemoryTracking struct ThreadState { - ThreadState(uint32_t workgroupIndex, Debugger &debugger, const GlobalState &globalState, - uint32_t maxSSAId); + ThreadState(Debugger &debugger, const GlobalState &globalState, uint32_t maxSSAId); ~ThreadState(); void EnterFunction(const DXIL::Function *function, const rdcarray &args); void EnterEntryPoint(const DXIL::Function *function, ShaderDebugState *state); void StepNext(ShaderDebugState *state, DebugAPIWrapper *apiWrapper, - const rdcarray &workgroups); + const rdcarray &workgroup); void StepOverNopInstructions(); bool Finished() const; bool InUniformBlock() const; - bool ExecuteInstruction(DebugAPIWrapper *apiWrapper, const rdcarray &workgroups); + bool ExecuteInstruction(DebugAPIWrapper *apiWrapper, const rdcarray &workgroup); void MarkResourceAccess(const rdcstr &name, const ResourceReferenceInfo &resRefInfo, bool directAccess, const ShaderDirectAccess &access, @@ -259,21 +258,22 @@ struct ThreadState void UpdateBackingMemoryFromVariable(void *ptr, uint64_t &allocSize, const ShaderVariable &var); void UpdateMemoryVariableFromBackingMemory(Id memoryId, const void *ptr); - void PerformGPUResourceOp(const rdcarray &workgroups, DXIL::Operation opCode, + void PerformGPUResourceOp(const rdcarray &workgroup, DXIL::Operation opCode, DXIL::DXOp dxOpCode, const ResourceReferenceInfo &resRef, DebugAPIWrapper *apiWrapper, const DXIL::Instruction &inst, ShaderVariable &result); void Sub(const ShaderVariable &a, const ShaderVariable &b, ShaderValue &ret) const; ShaderValue DDX(bool fine, DXIL::Operation opCode, DXIL::DXOp dxOpCode, - const rdcarray &quad, const DXIL::Value *dxilValue) const; + const rdcarray &workgroup, const DXIL::Value *dxilValue) const; ShaderValue DDY(bool fine, DXIL::Operation opCode, DXIL::DXOp dxOpCode, - const rdcarray &quad, const DXIL::Value *dxilValue) const; + const rdcarray &workgroup, const DXIL::Value *dxilValue) const; void ProcessScopeChange(const rdcarray &oldLive, const rdcarray &newLive); - void InitialiseHelper(const ThreadState &activeState); - static bool ThreadsAreDiverged(const rdcarray &workgroups); + static bool WorkgroupIsDiverged(const rdcarray &workgroup); + static bool QuadIsDiverged(const rdcarray &workgroup, + const rdcfixedarray &quadNeighbours); bool GetShaderVariableHelper(const DXIL::Value *dxilValue, DXIL::Operation op, DXIL::DXOp dxOpCode, ShaderVariable &var, bool flushDenormInput, bool isLive) const; @@ -288,9 +288,9 @@ struct ThreadState struct { - uint32_t coverage; - uint32_t primID; - uint32_t isFrontFace; + uint32_t coverage = ~0U; + uint32_t primID = ~0U; + uint32_t isFrontFace = false; } m_Semantics; Debugger &m_Debugger; @@ -326,13 +326,13 @@ struct ThreadState MemoryTracking m_Memory; // The instruction index within the current function - uint32_t m_FunctionInstructionIdx = ~0U; + uint32_t m_FunctionInstructionIdx = 0; const DXIL::Instruction *m_CurrentInstruction = NULL; // The current and previous function basic block index uint32_t m_Block = ~0U; uint32_t m_PreviousBlock = ~0U; // The global PC of the active instruction that was or will be executed on the current simulation step - uint32_t m_ActiveGlobalInstructionIdx = ~0U; + uint32_t m_ActiveGlobalInstructionIdx = 0; // SSA Ids guaranteed to be greater than 0 and less than this value uint32_t m_MaxSSAId; @@ -340,10 +340,17 @@ struct ThreadState rdcarray m_accessedSRVs; rdcarray m_accessedUAVs; - // index in the pixel quad + // quad ID (arbitrary, just used to find neighbours for derivatives) + uint32_t m_QuadId = 0; + // index in the pixel quad (relative to the active lane) + uint32_t m_QuadLaneIndex = ~0U; + // the lane indices of our quad neighbours + rdcfixedarray m_QuadNeighbours = {~0U, ~0U, ~0U, ~0U}; + // index in the workgroup uint32_t m_WorkgroupIndex = ~0U; - bool m_Killed = true; - bool m_Ended = true; + bool m_Dead = false; + bool m_Ended = false; + bool m_Helper = false; }; struct GlobalState @@ -408,16 +415,13 @@ struct GlobalState rdcarray constantBlocks; rdcarray constantBlocksData; - // workgroup private variables - rdcarray workgroups; - // resources may be read-write but the variable itself doesn't change rdcarray readOnlyResources; rdcarray readWriteResources; rdcarray samplers; - // Globals across workgroups including inputs (immutable) and outputs (mutable) + // Globals across workgroup including inputs (immutable) and outputs (mutable) rdcarray globals; - // Constants across workgroups + // Constants across workgroup rdcarray constants; // Memory created for global variables MemoryTracking memory; @@ -511,17 +515,48 @@ struct TypeData bool colMajorMat = false; }; +enum class ThreadProperty : uint32_t +{ + Helper, + QuadId, + QuadLane, + Active, + SubgroupId, + Count, +}; + +struct ThreadProperties +{ + rdcfixedarray()> props; + + uint32_t &operator[](ThreadProperty p) + { + if(p >= ThreadProperty::Count) + return props[0]; + return props[(uint32_t)p]; + } + + uint32_t operator[](ThreadProperty p) const + { + if(p >= ThreadProperty::Count) + return 0; + return props[(uint32_t)p]; + } +}; + class Debugger : public DXBCContainerDebugger { public: Debugger() : DXBCContainerDebugger(true){}; ShaderDebugTrace *BeginDebug(uint32_t eventId, const DXBC::DXBCContainer *dxbcContainer, - const ShaderReflection &reflection, uint32_t activeLaneIndex); + const ShaderReflection &reflection, uint32_t activeLaneIndex, + uint32_t workgroupSize); + void InitialiseWorkgroup(const rdcarray &workgroupProperties); rdcarray ContinueDebug(DebugAPIWrapper *apiWrapper); GlobalState &GetGlobalState() { return m_GlobalState; } - ThreadState &GetActiveLane() { return m_Workgroups[m_ActiveLaneIndex]; } - ThreadState &GetWorkgroup(const uint32_t i) { return m_Workgroups[i]; } - rdcarray &GetWorkgroups() { return m_Workgroups; } + ThreadState &GetActiveLane() { return m_Workgroup[m_ActiveLaneIndex]; } + ThreadState &GetWorkgroup(const uint32_t i) { return m_Workgroup[i]; } + rdcarray &GetWorkgroup() { return m_Workgroup; } const rdcarray &GetLiveGlobals() { return m_LiveGlobals; } static rdcstr GetResourceReferenceName(const DXIL::Program *program, DXIL::ResourceClass resClass, const BindingSlot &slot); @@ -544,7 +579,7 @@ private: void AddLocalVariable(const DXIL::SourceMappingInfo &srcMapping, uint32_t instructionIndex); void ParseDebugData(); - rdcarray m_Workgroups; + rdcarray m_Workgroup; std::map m_FunctionInfos; // the live mutable global variables, to initialise a stack frame's live list