diff --git a/renderdoc/driver/d3d12/d3d12_replay.cpp b/renderdoc/driver/d3d12/d3d12_replay.cpp index acef8b03b..4fb0a4747 100644 --- a/renderdoc/driver/d3d12/d3d12_replay.cpp +++ b/renderdoc/driver/d3d12/d3d12_replay.cpp @@ -2594,6 +2594,11 @@ rdcarray D3D12Replay::GetDescriptorAccess(uint32_t eventId) ret = pipe->staticDescriptorAccess; + const D3D12DynamicShaderFeedback &usage = m_BindlessFeedback.Usage[eventId]; + + if(usage.valid) + ret.append(usage.access); + WrappedID3D12DescriptorHeap *resourceHeap = NULL; WrappedID3D12DescriptorHeap *samplerHeap = NULL; for(ResourceId id : rs.heaps) @@ -2609,6 +2614,17 @@ rdcarray D3D12Replay::GetDescriptorAccess(uint32_t eventId) for(DescriptorAccess &access : ret) { + if(access.type == DescriptorType::Sampler) + access.descriptorStore = + samplerHeap ? rm->GetOriginalID(samplerHeap->GetResourceID()) : ResourceId(); + else + access.descriptorStore = + resourceHeap ? rm->GetOriginalID(resourceHeap->GetResourceID()) : ResourceId(); + + // for direct heap access, don't do anything more + if(access.index == DescriptorAccess::NoShaderBinding) + continue; + const D3D12RenderState::RootSignature &rootSig = pipe->IsGraphics() ? rs.graphics : rs.compute; uint32_t rootIndex = (uint32_t)access.byteSize; @@ -2623,13 +2639,6 @@ rdcarray D3D12Replay::GetDescriptorAccess(uint32_t eventId) continue; } - if(access.type == DescriptorType::Sampler) - access.descriptorStore = - samplerHeap ? rm->GetOriginalID(samplerHeap->GetResourceID()) : ResourceId(); - else - access.descriptorStore = - resourceHeap ? rm->GetOriginalID(resourceHeap->GetResourceID()) : ResourceId(); - const D3D12RenderState::SignatureElement &rootEl = rootSig.sigelems[rootIndex]; // this indicates a root parameter @@ -2648,163 +2657,6 @@ rdcarray D3D12Replay::GetDescriptorAccess(uint32_t eventId) access.byteOffset += (uint32_t)rootEl.offset; } } - - const D3D12DynamicShaderFeedback &usage = m_BindlessFeedback.Usage[eventId]; - - // decode dynamic usage by reverse looking up shader bindpoint mappings. This is a temporary - // measure, once the old style bindings reporting are removed we can refactor the shader - // feedback to provide our data more directly in the format we want - if(usage.valid) - { - const D3D12RenderState::RootSignature &rootSigBind = - pipe->IsGraphics() ? rs.graphics : rs.compute; - WrappedID3D12RootSignature *rootSig = - rm->GetCurrentAs(rootSigBind.rootsig); - - for(const D3D12FeedbackBindIdentifier &bind : usage.used) - { - DescriptorAccess access; - - if(bind.directAccess) - { - access.index = DescriptorAccess::NoShaderBinding; - access.arrayElement = bind.descIndex; - - switch(bind.bindType) - { - case BindType::Unknown: - case BindType::ImageSampler: - case BindType::InputAttachment: - default: - RDCERR("Unexpected current descriptor type referenced"); - access.type = DescriptorType::Unknown; - break; - case BindType::ConstantBuffer: access.type = DescriptorType::ConstantBuffer; break; - case BindType::Sampler: access.type = DescriptorType::Sampler; break; - case BindType::ReadOnlyImage: access.type = DescriptorType::Image; break; - case BindType::ReadWriteImage: access.type = DescriptorType::ReadWriteImage; break; - case BindType::ReadOnlyTBuffer: access.type = DescriptorType::TypedBuffer; break; - case BindType::ReadWriteTBuffer: - access.type = DescriptorType::ReadWriteTypedBuffer; - break; - case BindType::ReadOnlyBuffer: access.type = DescriptorType::Buffer; - case BindType::ReadWriteBuffer: access.type = DescriptorType::ReadWriteBuffer; break; - case BindType::ReadOnlyResource: access.type = DescriptorType::Image; break; - case BindType::ReadWriteResource: access.type = DescriptorType::ReadWriteImage; break; - } - - // don't have stage-access information here yet - access.stage = pipe->IsGraphics() ? ShaderStage::Pixel : ShaderStage::Compute; - - access.byteOffset = bind.descIndex; - } - else - { - if(bind.rootEl >= rootSig->sig.Parameters.size() || - bind.rangeIndex >= rootSig->sig.Parameters[bind.rootEl].ranges.size()) - { - RDCERR("Out-of-bounds root element referenced in dynamic usage"); - continue; - } - - const D3D12_DESCRIPTOR_RANGE1 &range = - rootSig->sig.Parameters[bind.rootEl].ranges[bind.rangeIndex]; - UINT space = range.RegisterSpace; - UINT reg = range.BaseShaderRegister + bind.descIndex; - - uint32_t prevRangeOffset = 0; - for(uint32_t prevRange = 0; prevRange < bind.rangeIndex; prevRange) - { - uint32_t rangeOffset = - rootSig->sig.Parameters[bind.rootEl].ranges[prevRange].OffsetInDescriptorsFromTableStart; - if(rangeOffset == D3D12_DESCRIPTOR_RANGE_OFFSET_APPEND) - rangeOffset = prevRangeOffset; - prevRangeOffset += rootSig->sig.Parameters[bind.rootEl].ranges[prevRange].NumDescriptors; - } - - bool found = false; - // this could have come from any stage, so we just find the first match - for(WrappedID3D12PipelineState::ShaderEntry *stage : - {pipe->VS(), pipe->HS(), pipe->DS(), pipe->GS(), pipe->PS(), pipe->CS(), pipe->AS(), - pipe->MS()}) - { - if(!stage) - continue; - - const rdcarray *iface = NULL; - switch(range.RangeType) - { - default: - case D3D12_DESCRIPTOR_RANGE_TYPE_SRV: - iface = &stage->GetMapping().readOnlyResources; - access.type = DescriptorType::Image; // hack - break; - case D3D12_DESCRIPTOR_RANGE_TYPE_UAV: - iface = &stage->GetMapping().readWriteResources; - access.type = DescriptorType::ReadWriteBuffer; // hack - break; - case D3D12_DESCRIPTOR_RANGE_TYPE_CBV: - iface = &stage->GetMapping().constantBlocks; - access.type = DescriptorType::ConstantBuffer; - break; - case D3D12_DESCRIPTOR_RANGE_TYPE_SAMPLER: - iface = &stage->GetMapping().samplers; - access.type = DescriptorType::Sampler; - break; - } - - if(iface) - { - access.index = 0; - for(const Bindpoint &searchBind : (*iface)) - { - if((uint32_t)searchBind.bindset == space && (uint32_t)searchBind.bind <= reg && - reg < uint32_t(searchBind.bind + searchBind.arraySize)) - { - access.stage = stage->GetDetails().stage; - access.arrayElement = reg - searchBind.bind; - access.byteOffset = range.OffsetInDescriptorsFromTableStart; - if(range.OffsetInDescriptorsFromTableStart == D3D12_DESCRIPTOR_RANGE_OFFSET_APPEND) - access.byteOffset = prevRangeOffset; - access.byteOffset += searchBind.bind - range.BaseShaderRegister; - access.byteOffset += access.arrayElement; - found = true; - break; - } - access.index++; - } - - if(found) - break; - } - } - } - - if(access.type == DescriptorType::Sampler) - access.descriptorStore = - samplerHeap ? rm->GetOriginalID(samplerHeap->GetResourceID()) : ResourceId(); - else - access.descriptorStore = - resourceHeap ? rm->GetOriginalID(resourceHeap->GetResourceID()) : ResourceId(); - access.byteSize = 1; - - // don't add any duplicates if there's already static descriptor access listed for this - // register - currently we report duplicates of static accesses in the dynamic feedback - bool found = false; - for(const DescriptorAccess &a : ret) - { - if(access.stage == a.stage && - CategoryForDescriptorType(access.type) == CategoryForDescriptorType(a.type) && - access.index == a.index && access.arrayElement == a.arrayElement) - { - found = true; - break; - } - } - if(!found) - ret.push_back(access); - } - } } return ret; diff --git a/renderdoc/driver/d3d12/d3d12_replay.h b/renderdoc/driver/d3d12/d3d12_replay.h index 083a581e1..57aba4136 100644 --- a/renderdoc/driver/d3d12/d3d12_replay.h +++ b/renderdoc/driver/d3d12/d3d12_replay.h @@ -301,6 +301,7 @@ private: { bool compute = false, valid = false; rdcarray used; + rdcarray access; }; struct Feedback diff --git a/renderdoc/driver/d3d12/d3d12_resources.h b/renderdoc/driver/d3d12/d3d12_resources.h index 8844fc8bb..e45874c01 100644 --- a/renderdoc/driver/d3d12/d3d12_resources.h +++ b/renderdoc/driver/d3d12/d3d12_resources.h @@ -29,6 +29,10 @@ #include "d3d12_device.h" #include "d3d12_manager.h" +rdcpair FindMatchingRootParameter(const D3D12RootSignature *sig, + D3D12_SHADER_VISIBILITY visibility, + D3D12_DESCRIPTOR_RANGE_TYPE rangeType, + uint32_t space, uint32_t bind); UINT GetPlaneForSubresource(ID3D12Resource *res, int Subresource); UINT GetMipForSubresource(ID3D12Resource *res, int Subresource); UINT GetSliceForSubresource(ID3D12Resource *res, int Subresource); diff --git a/renderdoc/driver/d3d12/d3d12_shader_feedback.cpp b/renderdoc/driver/d3d12/d3d12_shader_feedback.cpp index 925518730..8def11227 100644 --- a/renderdoc/driver/d3d12/d3d12_shader_feedback.cpp +++ b/renderdoc/driver/d3d12/d3d12_shader_feedback.cpp @@ -41,15 +41,26 @@ RDOC_CONFIG( struct D3D12FeedbackKey { + ShaderStage stage; DXBCBytecode::OperandType type; - Bindpoint bind; + uint32_t space; + uint32_t bind; bool operator<(const D3D12FeedbackKey &key) const { + if(stage != key.stage) + return stage < key.stage; if(type != key.type) return type < key.type; + if(space != key.space) + return space < key.space; return bind < key.bind; } + + bool operator==(const D3D12FeedbackKey &key) const + { + return stage == key.stage && type == key.type && space == key.space && bind == key.bind; + } }; struct D3D12FeedbackSlot @@ -64,24 +75,28 @@ public: void SetStaticUsed() { used = 0x1; } bool StaticUsed() const { return used != 0x0; } uint32_t Slot() const { return slot; } + + uint32_t numDescriptors = 1; + + DescriptorAccess access; private: uint32_t slot : 31; uint32_t used : 1; }; static const uint32_t numReservedSlots = 4; -static const uint32_t magicFeedbackValue = 0xbeebf330; -// keep the 4 lower bits to store the type of resource for direct heap access bindless -static const uint32_t magicFeedbackTypeMask = 0x0000000f; +static const uint32_t magicFeedbackValue = 0xfeed0000; +// the 8 lower bits to store the descriptor type of resource for direct heap access bindless +static const uint32_t magicFeedbackTypeMask = 0x000000ff; +// the next 8 bits indicate the shader stage mask of usage +static const uint32_t magicFeedbackStageMask = 0x0000ff00; +static const uint32_t magicFeedbackStageShift = 8; -static D3D12FeedbackKey GetDirectHeapAccessKey() +static D3D12FeedbackKey GetDirectHeapAccessKey(bool samplers) { - Bindpoint bind; - bind.bind = -1; - bind.bindset = -1; - D3D12FeedbackKey key; - key.type = DXBCBytecode::OperandType::TYPE_RESOURCE; - key.bind = bind; + D3D12FeedbackKey key = {}; + key.type = DXBCBytecode::OperandType::TYPE_THIS_POINTER; + key.space = samplers ? 1 : 0; return key; } @@ -157,17 +172,18 @@ static bool AnnotateDXBCShader(const DXBC::DXBCContainer *dxbc, uint32_t space, idx = imm(uint32_t(operand.indices[1].index - decl->operand.indices[1].index)); } - D3D12FeedbackKey key; + D3D12FeedbackKey key = {}; + key.stage = GetShaderStage(editor.GetShaderType()); key.type = operand.type; - key.bind.bindset = decl->space; - key.bind.bind = (int32_t)decl->operand.indices[1].index; + key.space = decl->space; + key.bind = (int32_t)decl->operand.indices[1].index; auto it = slots.find(key); if(it == slots.end()) { RDCERR("Couldn't find reserved base slot for %d at space %u and bind %u", key.type, - key.bind.bindset, key.bind.bind); + key.space, key.bind); continue; } @@ -250,7 +266,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, // register IDs of each SRV/UAV with slots, and record the base slot in this array for easy access // later when annotating dx.op.createHandle calls. We also need to know the base register because // the index dxc provides is register-relative - rdcarray> srvBaseSlots, uavBaseSlots; + rdcarray> srvBaseSlots, uavBaseSlots; // declare the resource, this happens purely in metadata but we need to store the slot uint32_t regSlot = 0; @@ -275,6 +291,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, uavs = reslist->children[1] = editor.CreateMetadata(); D3D12FeedbackKey key; + key.stage = GetShaderStage(editor.GetShaderType()); key.type = DXBCBytecode::TYPE_RESOURCE; @@ -305,8 +322,8 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, } uint32_t id = slot->getU32(); - key.bind.bindset = srvSpace->getU32(); - key.bind.bind = reg->getU32(); + key.space = srvSpace->getU32(); + key.bind = reg->getU32(); // ensure every valid ID has an index, even if it's 0 srvBaseSlots.resize_for_index(id); @@ -327,7 +344,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, // identifier for 'this resource isn't annotated' RDCASSERT(feedbackSlot > 0); - srvBaseSlots[id] = {feedbackSlot, key.bind.bind}; + srvBaseSlots[id] = {feedbackSlot, key.bind}; } key.type = DXBCBytecode::TYPE_UNORDERED_ACCESS_VIEW; @@ -367,8 +384,8 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, // ensure every valid ID has an index, even if it's 0 uavBaseSlots.resize_for_index(id); - key.bind.bindset = uavSpace->getU32(); - key.bind.bind = reg->getU32(); + key.space = uavSpace->getU32(); + key.bind = reg->getU32(); auto it = slots.find(key); @@ -386,7 +403,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, // identifier for 'this resource isn't annotated' RDCASSERT(feedbackSlot > 0); - uavBaseSlots[id] = {feedbackSlot, key.bind.bind}; + uavBaseSlots[id] = {feedbackSlot, key.bind}; } Constant rwundef; @@ -596,7 +613,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, (createHandleFromBinding && inst.getFuncCall()->name == createHandleFromBinding->name))) { Value *idxArg; - rdcpair slotInfo = {0, 0}; + rdcpair slotInfo = {0, 0}; if((createHandle && inst.getFuncCall()->name == createHandle->name)) { RDCASSERT(!isShaderModel6_6OrAbove); @@ -644,11 +661,12 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, } D3D12FeedbackKey key; + key.stage = GetShaderStage(editor.GetShaderType()); if(resBindArg->isNULL()) { key.type = DXBCBytecode::TYPE_RESOURCE; - key.bind.bindset = 0; - key.bind.bind = 0; + key.space = 0; + key.bind = 0; } else { @@ -666,8 +684,8 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, continue; key.type = kind == HandleKind::UAV ? DXBCBytecode::TYPE_UNORDERED_ACCESS_VIEW : DXBCBytecode::TYPE_RESOURCE; - key.bind.bindset = spaceArg->getU32(); - key.bind.bind = regArg->getU32(); + key.space = spaceArg->getU32(); + key.bind = regArg->getU32(); } auto it = slots.find(key); @@ -677,7 +695,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, // static used i.e. not arrayed? ignore if(it->second.StaticUsed()) continue; - slotInfo = {it->second.Slot(), key.bind.bind}; + slotInfo = {it->second.Slot(), key.bind}; } if(slotInfo.first == 0) continue; @@ -758,7 +776,7 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, } bool isSampler = isSamplerArg->getU32() != 0; - D3D12FeedbackKey key = GetDirectHeapAccessKey(); + D3D12FeedbackKey key = GetDirectHeapAccessKey(isSampler); auto it = slots.find(key); if(it == slots.end()) continue; @@ -807,29 +825,37 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, RDCERR("Unexpected non-constant argument for dx.types.ResourceProperties's resource kind"); continue; } - HandleKind handleKind = HandleKind::SRV; + DescriptorType descriptorType = DescriptorType::Unknown; ResourceKind resKind = (ResourceKind)(resKindArg->getU32() & 0xFF); bool isUav = (resKindArg->getU32() & (1 << 12)) != 0; if(resKind == ResourceKind::Sampler || resKind == ResourceKind::SamplerComparison) { - handleKind = HandleKind::Sampler; + descriptorType = DescriptorType::Sampler; RDCASSERT(isSampler); RDCASSERT(!isUav); } else if(resKind == ResourceKind::CBuffer) { - handleKind = HandleKind::CBuffer; + descriptorType = DescriptorType::ConstantBuffer; RDCASSERT(!isSampler); RDCASSERT(!isUav); } else if(isUav) { - handleKind = HandleKind::UAV; + descriptorType = DescriptorType::ReadWriteImage; + if(resKind == ResourceKind::TBuffer) + descriptorType = DescriptorType::ReadWriteTypedBuffer; + else if(resKind == ResourceKind::RawBuffer || resKind == ResourceKind::StructuredBuffer) + descriptorType = DescriptorType::ReadWriteBuffer; RDCASSERT(!isSampler); } else { - handleKind = HandleKind::SRV; + descriptorType = DescriptorType::Image; + if(resKind == ResourceKind::TBuffer || resKind == ResourceKind::TypedBuffer) + descriptorType = DescriptorType::TypedBuffer; + else if(resKind == ResourceKind::RawBuffer || resKind == ResourceKind::StructuredBuffer) + descriptorType = DescriptorType::Buffer; RDCASSERT(!isSampler); } @@ -867,7 +893,13 @@ static bool AnnotateDXILShader(const DXBC::DXBCContainer *dxbc, uint32_t space, editor.CreateConstant(2U), })); - uint32_t feedbackValue = magicFeedbackValue | (1 << (uint32_t)handleKind); + // this would in theory alias if there were different stages that used the same descriptor + // with different types, but we assume that won't happen (since that would mean texture + // reinterpreted as buffer or something invalid, not just 2D <-> 2DArray or some valid alias) + uint32_t feedbackValue = magicFeedbackValue; + feedbackValue |= ((uint32_t)descriptorType) & magicFeedbackTypeMask; + uint32_t stageMask = (uint32_t)MaskForStage(GetShaderStage(editor.GetShaderType())); + feedbackValue |= (stageMask << magicFeedbackStageShift) & magicFeedbackStageMask; editor.InsertInstruction(f, i++, editor.CreateInstruction(atomicBinOp, DXOp::atomicBinOp, { @@ -901,22 +933,32 @@ static bool AddArraySlots(WrappedID3D12PipelineState::ShaderEntry *shad, uint32_ bool dynamicUsed = false; ShaderReflection &refl = shad->GetDetails(); - const ShaderBindpointMapping &mapping = shad->GetMapping(); - for(const ShaderResource &ro : refl.readOnlyResources) + for(size_t i = 0; i < refl.readOnlyResources.size(); i++) { - const Bindpoint &bind = mapping.readOnlyResources[ro.bindPoint]; - if(bind.arraySize > 1) + const ShaderResource &ro = refl.readOnlyResources[i]; + if(ro.bindArraySize > 1) { D3D12FeedbackKey key; + key.stage = refl.stage; key.type = DXBCBytecode::TYPE_RESOURCE; - key.bind = bind; + key.space = ro.fixedBindSetOrSpace; + key.bind = ro.fixedBindNumber; + + DescriptorAccess access; + access.stage = refl.stage; + access.type = ro.descriptorType; + access.index = i & 0xffff; + // descriptor storage side will be calculated later when finalising this with the root + // signature information. slots[key].SetSlot(numSlots); - numSlots += RDCMIN(maxDescriptors, bind.arraySize); + slots[key].numDescriptors = RDCMIN(maxDescriptors, ro.bindArraySize); + slots[key].access = access; + numSlots += slots[key].numDescriptors; dynamicUsed = true; } - else if(bind.arraySize <= 1 && bind.used) + else if(ro.bindArraySize <= 1) { // since the eventual descriptor range iteration won't know which descriptors map to arrays // and which to fixed slots, it can't mark fixed descriptors as dynamically used itself. So @@ -924,27 +966,40 @@ static bool AddArraySlots(WrappedID3D12PipelineState::ShaderEntry *shad, uint32_ // they're fixed used. This allows for overlap between an array and a fixed resource which is // allowed D3D12FeedbackKey key; + key.stage = refl.stage; key.type = DXBCBytecode::TYPE_RESOURCE; - key.bind = bind; + key.space = ro.fixedBindSetOrSpace; + key.bind = ro.fixedBindNumber; slots[key].SetStaticUsed(); } } - for(const ShaderResource &rw : refl.readWriteResources) + for(size_t i = 0; i < refl.readWriteResources.size(); i++) { - const Bindpoint &bind = mapping.readWriteResources[rw.bindPoint]; - if(bind.arraySize > 1) + const ShaderResource &rw = refl.readWriteResources[i]; + if(rw.bindArraySize > 1) { D3D12FeedbackKey key; + key.stage = refl.stage; key.type = DXBCBytecode::TYPE_UNORDERED_ACCESS_VIEW; - key.bind = bind; + key.space = rw.fixedBindSetOrSpace; + key.bind = rw.fixedBindNumber; + + DescriptorAccess access; + access.stage = refl.stage; + access.type = rw.descriptorType; + access.index = i & 0xffff; + // descriptor storage side will be calculated later when finalising this with the root + // signature information. slots[key].SetSlot(numSlots); - numSlots += RDCMIN(maxDescriptors, bind.arraySize); + slots[key].numDescriptors = RDCMIN(maxDescriptors, rw.bindArraySize); + slots[key].access = access; + numSlots += slots[key].numDescriptors; dynamicUsed = true; } - else if(bind.arraySize <= 1 && bind.used) + else if(rw.bindArraySize <= 1) { // since the eventual descriptor range iteration won't know which descriptors map to arrays // and which to fixed slots, it can't mark fixed descriptors as dynamically used itself. So @@ -952,8 +1007,10 @@ static bool AddArraySlots(WrappedID3D12PipelineState::ShaderEntry *shad, uint32_ // they're fixed used. This allows for overlap between an array and a fixed resource which is // allowed D3D12FeedbackKey key; + key.stage = refl.stage; key.type = DXBCBytecode::TYPE_UNORDERED_ACCESS_VIEW; - key.bind = bind; + key.space = rw.fixedBindSetOrSpace; + key.bind = rw.fixedBindNumber; slots[key].SetStaticUsed(); } @@ -969,9 +1026,20 @@ static bool AddArraySlots(WrappedID3D12PipelineState::ShaderEntry *shad, uint32_ if(directHeapAccess) { - D3D12FeedbackKey key = GetDirectHeapAccessKey(); - slots[key].SetSlot(numSlots); - numSlots += maxDescriptors; + // only one stage should allocate space for direct heap access as it's shared + D3D12FeedbackKey key = GetDirectHeapAccessKey(false); + if(slots.find(key) == slots.end()) + { + slots[key].SetSlot(numSlots); + slots[key].numDescriptors = maxDescriptors; + numSlots += maxDescriptors; + + key = GetDirectHeapAccessKey(true); + slots[key].SetSlot(numSlots); + slots[key].numDescriptors = D3D12_MAX_SHADER_VISIBLE_SAMPLER_HEAP_SIZE; + numSlots += D3D12_MAX_SHADER_VISIBLE_SAMPLER_HEAP_SIZE; + } + dynamicUsed = true; } @@ -1133,7 +1201,7 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) D3D12_EXPANDED_PIPELINE_STATE_STREAM_DESC pipeDesc; pipe->Fill(pipeDesc); - uint32_t space = 1; + uint32_t feedbackBufSpace = 1; uint32_t maxDescriptors = 0; for(ResourceId id : rs.heaps) @@ -1145,7 +1213,7 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) } RDCDEBUG("Clamping any unbounded ranges to %u descriptors", maxDescriptors); - std::map slots[(uint32_t)ShaderStage::Count]; + std::map slots; // reserve the first 4 dwords for debug info and a validity flag uint32_t numSlots = numReservedSlots; @@ -1170,11 +1238,11 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) bool directHeapAccess = (modsig.Flags & (D3D12_ROOT_SIGNATURE_FLAG_CBV_SRV_UAV_HEAP_DIRECTLY_INDEXED | D3D12_ROOT_SIGNATURE_FLAG_SAMPLER_HEAP_DIRECTLY_INDEXED)) != 0; - space = modsig.maxSpaceIndex; + feedbackBufSpace = modsig.maxSpaceIndex; - dynamicAccessPerStage[5] = AddArraySlots( - pipe->CS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Compute], numSlots, - editedBlob[(uint32_t)ShaderStage::Compute], pipeDesc.CS, directHeapAccess); + dynamicAccessPerStage[5] = + AddArraySlots(pipe->CS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[(uint32_t)ShaderStage::Compute], pipeDesc.CS, directHeapAccess); } else { @@ -1191,29 +1259,29 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) (modsig.Flags & (D3D12_ROOT_SIGNATURE_FLAG_CBV_SRV_UAV_HEAP_DIRECTLY_INDEXED | D3D12_ROOT_SIGNATURE_FLAG_SAMPLER_HEAP_DIRECTLY_INDEXED)) != 0; - space = modsig.maxSpaceIndex; + feedbackBufSpace = modsig.maxSpaceIndex; - dynamicAccessPerStage[0] = AddArraySlots( - pipe->VS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Vertex], numSlots, - editedBlob[uint32_t(ShaderStage::Vertex)], pipeDesc.VS, directHeapAccess); - dynamicAccessPerStage[1] = AddArraySlots( - pipe->HS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Hull], numSlots, - editedBlob[uint32_t(ShaderStage::Hull)], pipeDesc.HS, directHeapAccess); - dynamicAccessPerStage[2] = AddArraySlots( - pipe->DS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Domain], numSlots, - editedBlob[uint32_t(ShaderStage::Domain)], pipeDesc.DS, directHeapAccess); - dynamicAccessPerStage[3] = AddArraySlots( - pipe->GS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Geometry], numSlots, - editedBlob[uint32_t(ShaderStage::Geometry)], pipeDesc.GS, directHeapAccess); - dynamicAccessPerStage[4] = AddArraySlots( - pipe->PS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Pixel], numSlots, - editedBlob[uint32_t(ShaderStage::Pixel)], pipeDesc.PS, directHeapAccess); + dynamicAccessPerStage[0] = + AddArraySlots(pipe->VS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[uint32_t(ShaderStage::Vertex)], pipeDesc.VS, directHeapAccess); + dynamicAccessPerStage[1] = + AddArraySlots(pipe->HS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[uint32_t(ShaderStage::Hull)], pipeDesc.HS, directHeapAccess); + dynamicAccessPerStage[2] = + AddArraySlots(pipe->DS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[uint32_t(ShaderStage::Domain)], pipeDesc.DS, directHeapAccess); + dynamicAccessPerStage[3] = + AddArraySlots(pipe->GS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[uint32_t(ShaderStage::Geometry)], pipeDesc.GS, directHeapAccess); + dynamicAccessPerStage[4] = + AddArraySlots(pipe->PS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[uint32_t(ShaderStage::Pixel)], pipeDesc.PS, directHeapAccess); dynamicAccessPerStage[6] = AddArraySlots( - pipe->AS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Amplification], numSlots, + pipe->AS(), feedbackBufSpace, maxDescriptors, slots, numSlots, editedBlob[uint32_t(ShaderStage::Amplification)], pipeDesc.AS, directHeapAccess); - dynamicAccessPerStage[7] = AddArraySlots( - pipe->MS(), space, maxDescriptors, slots[(uint32_t)ShaderStage::Mesh], numSlots, - editedBlob[uint32_t(ShaderStage::Mesh)], pipeDesc.MS, directHeapAccess); + dynamicAccessPerStage[7] = + AddArraySlots(pipe->MS(), feedbackBufSpace, maxDescriptors, slots, numSlots, + editedBlob[uint32_t(ShaderStage::Mesh)], pipeDesc.MS, directHeapAccess); } // if numSlots wasn't increased, none of the resources were arrayed so we have nothing to do. @@ -1238,7 +1306,7 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) param.ParameterType = D3D12_ROOT_PARAMETER_TYPE_UAV; param.ShaderVisibility = D3D12_SHADER_VISIBILITY_ALL; param.Descriptor.Flags = D3D12_ROOT_DESCRIPTOR_FLAG_DATA_VOLATILE; - param.Descriptor.RegisterSpace = space; + param.Descriptor.RegisterSpace = feedbackBufSpace; param.Descriptor.ShaderRegister = 0; } @@ -1442,31 +1510,25 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) curIdentifier.rangeIndex = r; - curKey.bind.bindset = range.RegisterSpace; + curKey.space = range.RegisterSpace; UINT num = range.NumDescriptors; - uint32_t visMask = 0; + ShaderStageMask visMask = ShaderStageMask::Unknown; // see which shader's binds we should look up for this range switch(p.ShaderVisibility) { case D3D12_SHADER_VISIBILITY_ALL: - visMask = result.compute ? (uint32_t)ShaderStageMask::Compute : 0xff; + visMask = result.compute ? ShaderStageMask::Compute : ShaderStageMask::All; break; - case D3D12_SHADER_VISIBILITY_VERTEX: - visMask = (uint32_t)ShaderStageMask::Vertex; - break; - case D3D12_SHADER_VISIBILITY_HULL: visMask = (uint32_t)ShaderStageMask::Hull; break; - case D3D12_SHADER_VISIBILITY_DOMAIN: - visMask = (uint32_t)ShaderStageMask::Domain; - break; - case D3D12_SHADER_VISIBILITY_GEOMETRY: - visMask = (uint32_t)ShaderStageMask::Geometry; - break; - case D3D12_SHADER_VISIBILITY_PIXEL: visMask = uint32_t(ShaderStageMask::Pixel); break; + case D3D12_SHADER_VISIBILITY_VERTEX: visMask = ShaderStageMask::Vertex; break; + case D3D12_SHADER_VISIBILITY_HULL: visMask = ShaderStageMask::Hull; break; + case D3D12_SHADER_VISIBILITY_DOMAIN: visMask = ShaderStageMask::Domain; break; + case D3D12_SHADER_VISIBILITY_GEOMETRY: visMask = ShaderStageMask::Geometry; break; + case D3D12_SHADER_VISIBILITY_PIXEL: visMask = ShaderStageMask::Pixel; break; case D3D12_SHADER_VISIBILITY_AMPLIFICATION: - visMask = uint32_t(ShaderStageMask::Amplification); + visMask = ShaderStageMask::Amplification; break; - case D3D12_SHADER_VISIBILITY_MESH: visMask = uint32_t(ShaderStageMask::Mesh); break; + case D3D12_SHADER_VISIBILITY_MESH: visMask = ShaderStageMask::Mesh; break; default: RDCERR("Unexpected shader visibility %d", p.ShaderVisibility); return true; } @@ -1476,58 +1538,60 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) else if(range.RangeType == D3D12_DESCRIPTOR_RANGE_TYPE_UAV) curKey.type = DXBCBytecode::TYPE_UNORDERED_ACCESS_VIEW; - for(uint32_t st = 0; st < (uint32_t)ShaderStage::Count; st++) + for(ShaderStage st : values()) { - if(visMask & (1 << st)) + curKey.stage = st; + if(visMask & MaskForStage(st)) { curIdentifier.descIndex = 0; - curKey.bind.bind = range.BaseShaderRegister; + curKey.bind = range.BaseShaderRegister; // the feedback entries start here - auto slotIt = slots[st].lower_bound(curKey); + auto slotIt = slots.lower_bound(curKey); // iterate over the declared range. This could be unbounded, so we might exit // another way for(uint32_t i = 0; i < num; i++) { // stop when we've run out of recorded used slots - if(slotIt == slots[st].end()) + if(slotIt == slots.end()) break; - Bindpoint bind = slotIt->first.bind; + uint32_t space = slotIt->first.space; + uint32_t bind = slotIt->first.bind; + uint32_t numDescriptors = slotIt->second.numDescriptors; // stop if the next used slot is in another space or is another type - if(bind.bindset > curKey.bind.bindset || slotIt->first.type != curKey.type) + if(space > curKey.space || slotIt->first.type != curKey.type) break; // if the next bind is definitely outside this range, early out now instead of // iterating fruitlessly - if(((uint32_t)bind.bind - range.BaseShaderRegister) > num) + if((bind - range.BaseShaderRegister) > num) break; - int32_t lastBind = - bind.bind + (int32_t)RDCCLAMP(bind.arraySize, 1U, maxDescriptors); + uint32_t lastBind = bind + RDCCLAMP(numDescriptors, 1U, maxDescriptors); // if this slot's array covers the current bind, check the result - if(bind.bind <= curKey.bind.bind && curKey.bind.bind < lastBind) + if(bind <= curKey.bind && curKey.bind < lastBind) { // if it's static used by having a fixed result declared, it's used const bool staticUsed = slotIt->second.StaticUsed(); // otherwise check the feedback we got const uint32_t baseSlot = slotIt->second.Slot(); - const uint32_t arrayIndex = curKey.bind.bind - bind.bind; + const uint32_t arrayIndex = curKey.bind - bind; if(staticUsed || slotsData[baseSlot + arrayIndex]) result.used.push_back(curIdentifier); } - curKey.bind.bind++; + curKey.bind++; curIdentifier.descIndex++; // if we've passed this slot, move to the next one. Because we're iterating a // contiguous range of binds the next slot will be enough for the next iteration - if(curKey.bind.bind >= lastBind) + if(curKey.bind >= lastBind) slotIt++; } } @@ -1536,51 +1600,155 @@ bool D3D12Replay::FetchShaderFeedback(uint32_t eventId) } } - D3D12FeedbackKey directAccessKey = GetDirectHeapAccessKey(); - for(uint32_t shaderStage = 0; shaderStage < (uint32_t)ShaderStage::Count; shaderStage++) + for(D3D12FeedbackKey directAccessKey : + {GetDirectHeapAccessKey(false), GetDirectHeapAccessKey(true)}) { - if(slots[shaderStage].find(directAccessKey) == slots[shaderStage].end()) - continue; - D3D12FeedbackSlot &feedbackSlots = slots[shaderStage].at(directAccessKey); - for(uint32_t i = feedbackSlots.Slot(); i < feedbackSlots.Slot() + maxDescriptors; ++i) + auto slotIt = slots.find(directAccessKey); + if(slotIt != slots.end()) { - if((slotsData[i] & magicFeedbackValue) == magicFeedbackValue) + const D3D12FeedbackSlot &feedbackSlots = slots.at(directAccessKey); + for(uint32_t i = feedbackSlots.Slot(); + i < feedbackSlots.Slot() + feedbackSlots.numDescriptors; ++i) { - uint32_t usedSlot = i - feedbackSlots.Slot(); - D3D12FeedbackBindIdentifier directAccessIdentifier = {}; - directAccessIdentifier.descIndex = usedSlot; - directAccessIdentifier.rootEl = ~0U; - directAccessIdentifier.rangeIndex = ~0U; - directAccessIdentifier.directAccess = true; - directAccessIdentifier.shaderStage = (ShaderStage)shaderStage; - uint32_t handleKind = slotsData[i] & magicFeedbackTypeMask; - if((handleKind & (1 << (uint32_t)DXIL::HandleKind::Sampler)) != 0) + if((slotsData[i] & magicFeedbackValue) == magicFeedbackValue) { - directAccessIdentifier.bindType = BindType::Sampler; - result.used.push_back(directAccessIdentifier); - } - bool isCBV = (handleKind & (1 << (uint32_t)DXIL::HandleKind::CBuffer)) != 0; - bool isSRV = (handleKind & (1 << (uint32_t)DXIL::HandleKind::SRV)) != 0; - bool isUAV = (handleKind & (1 << (uint32_t)DXIL::HandleKind::UAV)) != 0; - if(isCBV || isSRV || isUAV) - { - if((isCBV && isSRV) || (isCBV && isUAV) || (isSRV && isUAV)) + uint32_t usedSlot = i - feedbackSlots.Slot(); + D3D12FeedbackBindIdentifier directAccessIdentifier = {}; + directAccessIdentifier.descIndex = usedSlot; + directAccessIdentifier.rootEl = ~0U; + directAccessIdentifier.rangeIndex = ~0U; + directAccessIdentifier.directAccess = true; + + ShaderStageMask feedbackStages = + ShaderStageMask((slotsData[i] & magicFeedbackStageMask) >> magicFeedbackStageShift); + + DescriptorType descriptorType = DescriptorType(slotsData[i] & magicFeedbackTypeMask); + + if(descriptorType == DescriptorType::Sampler) + { + directAccessIdentifier.bindType = BindType::Sampler; + } + else if(IsConstantBlockDescriptor(descriptorType)) + { + directAccessIdentifier.bindType = BindType::ConstantBuffer; + } + else if(IsReadOnlyDescriptor(descriptorType)) + { + directAccessIdentifier.bindType = BindType::ReadOnlyResource; + } + else if(IsReadWriteDescriptor(descriptorType)) + { + directAccessIdentifier.bindType = BindType::ReadWriteResource; + } + else { - RDCERR("Unexpected, resource used with multiple incompatible types"); continue; } - if(isCBV) - directAccessIdentifier.bindType = BindType::ConstantBuffer; - else if(isSRV) - directAccessIdentifier.bindType = BindType::ReadOnlyResource; - else if(isUAV) - directAccessIdentifier.bindType = BindType::ReadWriteResource; - result.used.push_back(directAccessIdentifier); + + // add a usage for each stage that reported it used + for(ShaderStage stage : values()) + { + if(MaskForStage(stage) & feedbackStages) + { + directAccessIdentifier.shaderStage = stage; + result.used.push_back(directAccessIdentifier); + } + } } } } } + D3D12FeedbackKey resourceDirectAccessKey = GetDirectHeapAccessKey(false), + samplerDirectAccessKey = GetDirectHeapAccessKey(true); + for(auto it = slots.begin(); it != slots.end(); ++it) + { + if(it->first == resourceDirectAccessKey || it->first == samplerDirectAccessKey) + { + for(uint32_t i = it->second.Slot(); i < it->second.Slot() + it->second.numDescriptors; ++i) + { + if((slotsData[i] & magicFeedbackValue) == magicFeedbackValue) + { + DescriptorAccess access; + + access.index = DescriptorAccess::NoShaderBinding; + access.byteOffset = access.arrayElement = i - it->second.Slot(); + access.byteSize = 1; + // descriptorStore will be set before this is returned by the replay, it's implicit + // from the type + + ShaderStageMask feedbackStages = + ShaderStageMask((slotsData[i] & magicFeedbackStageMask) >> magicFeedbackStageShift); + + access.type = DescriptorType(slotsData[i] & magicFeedbackTypeMask); + + // add a usage for each stage that reported it used + for(ShaderStage stage : values()) + { + if(MaskForStage(stage) & feedbackStages) + { + access.stage = stage; + result.access.push_back(access); + } + } + } + } + + continue; + } + + // ignore static used slots, we don't need these for the descriptor access representation + if(it->second.numDescriptors == 1) + continue; + + DescriptorAccess access = it->second.access; + + D3D12_SHADER_VISIBILITY visibility; + switch(access.stage) + { + case ShaderStage::Vertex: visibility = D3D12_SHADER_VISIBILITY_VERTEX; break; + case ShaderStage::Hull: visibility = D3D12_SHADER_VISIBILITY_HULL; break; + case ShaderStage::Domain: visibility = D3D12_SHADER_VISIBILITY_DOMAIN; break; + case ShaderStage::Geometry: visibility = D3D12_SHADER_VISIBILITY_GEOMETRY; break; + case ShaderStage::Pixel: visibility = D3D12_SHADER_VISIBILITY_PIXEL; break; + case ShaderStage::Amplification: + visibility = D3D12_SHADER_VISIBILITY_AMPLIFICATION; + break; + case ShaderStage::Mesh: visibility = D3D12_SHADER_VISIBILITY_MESH; break; + case ShaderStage::Compute: visibility = D3D12_SHADER_VISIBILITY_ALL; break; + case ShaderStage::Count: + default: RDCERR("Invalid shader stage"); continue; + } + + D3D12_DESCRIPTOR_RANGE_TYPE rangeType = D3D12_DESCRIPTOR_RANGE_TYPE_SRV; + if(IsConstantBlockDescriptor(access.type)) + rangeType = D3D12_DESCRIPTOR_RANGE_TYPE_CBV; + else if(IsConstantBlockDescriptor(access.type)) + rangeType = D3D12_DESCRIPTOR_RANGE_TYPE_SAMPLER; + else if(IsReadOnlyDescriptor(access.type)) + rangeType = D3D12_DESCRIPTOR_RANGE_TYPE_SRV; + else if(IsReadWriteDescriptor(access.type)) + rangeType = D3D12_DESCRIPTOR_RANGE_TYPE_UAV; + + for(uint32_t i = 0; i < it->second.numDescriptors; i++) + { + const uint32_t baseSlot = it->second.Slot(); + const uint32_t arrayIndex = i; + + if(slotsData[baseSlot + arrayIndex]) + { + access.arrayElement = i; + rdctie(access.byteSize, access.byteOffset) = FindMatchingRootParameter( + &modsig, visibility, rangeType, it->first.space, it->first.bind); + + access.byteOffset += access.arrayElement; + + if(access.byteSize != ~0U) + result.access.push_back(access); + } + } + } + std::sort(result.used.begin(), result.used.end()); } else if((pipelinestats->VSInvocations == 0 || !dynamicAccessPerStage[0]) && diff --git a/renderdoc/driver/shaders/dxbc/dxbc_common.h b/renderdoc/driver/shaders/dxbc/dxbc_common.h index 509e918ce..ec239fd85 100644 --- a/renderdoc/driver/shaders/dxbc/dxbc_common.h +++ b/renderdoc/driver/shaders/dxbc/dxbc_common.h @@ -76,6 +76,8 @@ enum class ShaderType Max, }; +ShaderStage GetShaderStage(ShaderType type); + ///////////////////////////////////////////////////////////////////////// // the below classes basically mimics the existing reflection interface. // diff --git a/renderdoc/driver/shaders/dxbc/dxbc_container.cpp b/renderdoc/driver/shaders/dxbc/dxbc_container.cpp index ac1ee2a14..dcd1a07f6 100644 --- a/renderdoc/driver/shaders/dxbc/dxbc_container.cpp +++ b/renderdoc/driver/shaders/dxbc/dxbc_container.cpp @@ -290,6 +290,22 @@ ShaderBuiltin GetSystemValue(SVSemantic systemValue) return ShaderBuiltin::Undefined; } +ShaderStage GetShaderStage(ShaderType type) +{ + switch(type) + { + case DXBC::ShaderType::Pixel: return ShaderStage::Pixel; + case DXBC::ShaderType::Vertex: return ShaderStage::Vertex; + case DXBC::ShaderType::Geometry: return ShaderStage::Geometry; + case DXBC::ShaderType::Hull: return ShaderStage::Hull; + case DXBC::ShaderType::Domain: return ShaderStage::Domain; + case DXBC::ShaderType::Compute: return ShaderStage::Compute; + case DXBC::ShaderType::Amplification: return ShaderStage::Amplification; + case DXBC::ShaderType::Mesh: return ShaderStage::Mesh; + default: RDCERR("Unexpected DXBC shader type %u", type); return ShaderStage::Vertex; + } +} + rdcstr TypeName(CBufferVariableType desc) { rdcstr ret;