mirror of
https://github.com/baldurk/renderdoc.git
synced 2026-08-27 00:46:36 +00:00
Update remaining I/O handling to support non-32-bit types
* This includes the shader input interface, and storage/cbuffer access.
This commit is contained in:
@@ -2697,6 +2697,8 @@ void ThreadState::StepNext(ShaderDebugState *state, const rdcarray<ThreadState>
|
||||
|
||||
ShaderVariable result;
|
||||
result.type = resultType.scalar().Type();
|
||||
result.rows = 1;
|
||||
result.columns = RDCMAX(1U, resultType.vector().count) & 0xff;
|
||||
|
||||
DebugAPIWrapper::TextureType texType = debugger.GetTextureType(img);
|
||||
|
||||
@@ -2735,9 +2737,6 @@ void ThreadState::StepNext(ShaderDebugState *state, const rdcarray<ThreadState>
|
||||
}
|
||||
}
|
||||
|
||||
result.rows = 1;
|
||||
result.columns = RDCMAX(1U, resultType.vector().count) & 0xff;
|
||||
|
||||
SetDst(read.result, result);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -92,14 +92,14 @@ public:
|
||||
|
||||
struct DerivativeDeltas
|
||||
{
|
||||
Vec4f ddxcoarse;
|
||||
Vec4f ddycoarse;
|
||||
Vec4f ddxfine;
|
||||
Vec4f ddyfine;
|
||||
ShaderVariable ddxcoarse;
|
||||
ShaderVariable ddycoarse;
|
||||
ShaderVariable ddxfine;
|
||||
ShaderVariable ddyfine;
|
||||
};
|
||||
|
||||
virtual DerivativeDeltas GetDerivative(ShaderBuiltin builtin, uint32_t location,
|
||||
uint32_t component) = 0;
|
||||
uint32_t component, VarType type) = 0;
|
||||
};
|
||||
|
||||
typedef ShaderVariable (*ExtInstImpl)(ThreadState &, uint32_t, const rdcarray<Id> &);
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
#include "common/formatting.h"
|
||||
#include "spirv_op_helpers.h"
|
||||
#include "spirv_reflect.h"
|
||||
#include "var_dispatch_helpers.h"
|
||||
|
||||
// this could be cleaner if ShaderVariable wasn't a very public struct, but it's not worth it so
|
||||
// we just reserve value slots that we know won't be used in opaque variables
|
||||
@@ -89,6 +90,20 @@ static uint32_t VarByteSize(const ShaderVariable &var)
|
||||
RDCMAX(1U, (uint32_t)var.columns);
|
||||
}
|
||||
|
||||
static void *VarElemPointer(ShaderVariable &var, uint32_t comp)
|
||||
{
|
||||
RDCASSERTNOTEQUAL(var.type, VarType::Unknown);
|
||||
byte *ret = (byte *)var.value.u64v;
|
||||
return ret + comp * VarTypeByteSize(var.type);
|
||||
}
|
||||
|
||||
static const void *VarElemPointer(const ShaderVariable &var, uint32_t comp)
|
||||
{
|
||||
RDCASSERTNOTEQUAL(var.type, VarType::Unknown);
|
||||
const byte *ret = (const byte *)var.value.u64v;
|
||||
return ret + comp * VarTypeByteSize(var.type);
|
||||
}
|
||||
|
||||
namespace rdcspv
|
||||
{
|
||||
void AssignValue(ShaderVariable &dst, const ShaderVariable &src)
|
||||
@@ -779,8 +794,7 @@ ShaderDebugTrace *Debugger::BeginDebug(DebugAPIWrapper *api, const ShaderStage s
|
||||
else
|
||||
{
|
||||
// matrix case is more complicated. Either read column by column or row by row
|
||||
// depending on
|
||||
// majorness
|
||||
// depending on majorness
|
||||
uint32_t matrixStride = curDecorations.matrixStride;
|
||||
|
||||
if(!(curDecorations.flags & Decorations::HasMatrixStride))
|
||||
@@ -791,18 +805,19 @@ ShaderDebugTrace *Debugger::BeginDebug(DebugAPIWrapper *api, const ShaderStage s
|
||||
|
||||
if(curDecorations.flags & Decorations::ColMajor)
|
||||
{
|
||||
ShaderValue tmp;
|
||||
ShaderVariable tmp;
|
||||
tmp.type = var.type;
|
||||
|
||||
uint32_t colSize = VarTypeByteSize(var.type) * var.rows;
|
||||
for(uint32_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
// read the column
|
||||
this->apiWrapper->ReadBufferValue(bindpoint, offset + c * matrixStride, colSize,
|
||||
&tmp.uv[0]);
|
||||
VarElemPointer(tmp, 0));
|
||||
|
||||
// now write it into the appropiate elements in the destination ShaderValue
|
||||
for(uint32_t r = 0; r < var.rows; r++)
|
||||
var.value.uv[r * var.columns + c] = tmp.uv[r];
|
||||
copyComp(var, r * var.columns + c, tmp, r, var.type);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -814,7 +829,7 @@ ShaderDebugTrace *Debugger::BeginDebug(DebugAPIWrapper *api, const ShaderStage s
|
||||
// read the column into the destination ShaderValue, which is tightly packed with
|
||||
// rows
|
||||
this->apiWrapper->ReadBufferValue(bindpoint, offset + r * matrixStride, rowSize,
|
||||
&var.value.uv[r * var.columns]);
|
||||
VarElemPointer(var, r * var.columns));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1493,30 +1508,26 @@ ShaderVariable Debugger::ReadFromPointer(const ShaderVariable &ptr) const
|
||||
{
|
||||
apiWrapper->ReadBufferValue(bind, offset + r * matrixStride,
|
||||
VarTypeByteSize(var.type) * var.columns,
|
||||
&var.value.uv[r * var.columns]);
|
||||
VarElemPointer(var, r * var.columns));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
ShaderValue tmp = {};
|
||||
ShaderVariable tmp;
|
||||
tmp.type = var.type;
|
||||
|
||||
// read column-wise
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
apiWrapper->ReadBufferValue(bind, offset + c * matrixStride,
|
||||
VarTypeByteSize(var.type) * var.rows, &tmp.uv[c * var.rows]);
|
||||
VarTypeByteSize(var.type) * var.rows,
|
||||
VarElemPointer(tmp, c * var.rows));
|
||||
}
|
||||
|
||||
// transpose into our row major storage
|
||||
for(uint8_t r = 0; r < var.rows; r++)
|
||||
{
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
if(VarTypeByteSize(var.type) == 8)
|
||||
var.value.u64v[r * var.columns + c] = tmp.u64v[c * var.rows + r];
|
||||
else
|
||||
var.value.uv[r * var.columns + c] = tmp.uv[c * var.rows + r];
|
||||
}
|
||||
}
|
||||
copyComp(var, r * var.columns + c, tmp, c * var.rows + r, var.type);
|
||||
}
|
||||
}
|
||||
else if(type.type == DataType::VectorType)
|
||||
@@ -1525,24 +1536,20 @@ ShaderVariable Debugger::ReadFromPointer(const ShaderVariable &ptr) const
|
||||
{
|
||||
// we can read a vector at a time if the matrix is column major
|
||||
apiWrapper->ReadBufferValue(bind, offset, VarTypeByteSize(var.type) * var.columns,
|
||||
var.value.uv);
|
||||
VarElemPointer(var, 0));
|
||||
}
|
||||
else
|
||||
{
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
if(VarTypeByteSize(var.type) == 8)
|
||||
apiWrapper->ReadBufferValue(bind, offset + c * matrixStride,
|
||||
VarTypeByteSize(var.type), &var.value.u64v[c]);
|
||||
else
|
||||
apiWrapper->ReadBufferValue(bind, offset + c * matrixStride,
|
||||
VarTypeByteSize(var.type), &var.value.uv[c]);
|
||||
apiWrapper->ReadBufferValue(bind, offset + c * matrixStride, VarTypeByteSize(var.type),
|
||||
VarElemPointer(var, VarTypeByteSize(var.type) * c));
|
||||
}
|
||||
}
|
||||
}
|
||||
else if(type.type == DataType::ScalarType)
|
||||
{
|
||||
apiWrapper->ReadBufferValue(bind, offset, VarTypeByteSize(var.type), var.value.uv);
|
||||
apiWrapper->ReadBufferValue(bind, offset, VarTypeByteSize(var.type), VarElemPointer(var, 0));
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1569,7 +1576,7 @@ ShaderVariable Debugger::ReadFromPointer(const ShaderVariable &ptr) const
|
||||
uint32_t scalar0 = (uint32_t)ptr.value.u64v[Scalar0VariableSlot];
|
||||
uint32_t scalar1 = (uint32_t)ptr.value.u64v[Scalar1VariableSlot];
|
||||
|
||||
ShaderValue val = {};
|
||||
ShaderVariable tmp = ret;
|
||||
|
||||
if(ret.rows > 1)
|
||||
{
|
||||
@@ -1580,29 +1587,19 @@ ShaderVariable Debugger::ReadFromPointer(const ShaderVariable &ptr) const
|
||||
{
|
||||
// two indices - selecting a scalar. scalar0 is the first index in the chain so it chooses
|
||||
// column
|
||||
if(VarTypeByteSize(ret.type) == 8)
|
||||
val.u64v[0] = ret.value.u64v[scalar1 * ret.columns + scalar0];
|
||||
else
|
||||
val.uv[0] = ret.value.uv[scalar1 * ret.columns + scalar0];
|
||||
copyComp(ret, 0, tmp, scalar1 * ret.columns + scalar0);
|
||||
|
||||
// it's a scalar now, even if it was a matrix before
|
||||
ret.rows = ret.columns = 1;
|
||||
ret.value = val;
|
||||
}
|
||||
else if(scalar0 != ~0U)
|
||||
{
|
||||
// one index, selecting a column
|
||||
for(uint32_t row = 0; row < ret.rows; row++)
|
||||
{
|
||||
if(VarTypeByteSize(ret.type) == 8)
|
||||
val.u64v[row] = ret.value.u64v[row * ret.columns + scalar0];
|
||||
else
|
||||
val.uv[row] = ret.value.uv[row * ret.columns + scalar0];
|
||||
}
|
||||
copyComp(ret, row, tmp, row * ret.columns + scalar0);
|
||||
|
||||
// it's a vector now, even if it was a matrix before
|
||||
ret.rows = 1;
|
||||
ret.value = val;
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -1612,14 +1609,10 @@ ShaderVariable Debugger::ReadFromPointer(const ShaderVariable &ptr) const
|
||||
// vector case, selecting a scalar (if anything)
|
||||
if(scalar0 != ~0U)
|
||||
{
|
||||
if(VarTypeByteSize(ret.type) == 8)
|
||||
val.u64v[0] = ret.value.u64v[scalar0];
|
||||
else
|
||||
val.uv[0] = ret.value.uv[scalar0];
|
||||
copyComp(ret, 0, tmp, scalar0);
|
||||
|
||||
// it's a scalar now, even if it was a matrix before
|
||||
ret.columns = 1;
|
||||
ret.value = val;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1687,30 +1680,24 @@ void Debugger::WriteThroughPointer(const ShaderVariable &ptr, const ShaderVariab
|
||||
{
|
||||
apiWrapper->WriteBufferValue(bind, offset + r * matrixStride,
|
||||
VarTypeByteSize(var.type) * var.columns,
|
||||
&var.value.uv[r * var.columns]);
|
||||
VarElemPointer(var, r * var.columns));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
ShaderValue tmp = {};
|
||||
ShaderVariable tmp;
|
||||
|
||||
// transpose from our row major storage
|
||||
for(uint8_t r = 0; r < var.rows; r++)
|
||||
{
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
if(VarTypeByteSize(var.type) == 8)
|
||||
tmp.u64v[c * var.rows + r] = var.value.u64v[r * var.columns + c];
|
||||
else
|
||||
tmp.uv[c * var.rows + r] = var.value.uv[r * var.columns + c];
|
||||
}
|
||||
}
|
||||
copyComp(tmp, c * var.rows + r, var, r * var.columns + c, var.type);
|
||||
|
||||
// read column-wise
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
apiWrapper->WriteBufferValue(bind, offset + c * matrixStride,
|
||||
VarTypeByteSize(var.type) * var.rows, &tmp.uv[c * var.rows]);
|
||||
VarTypeByteSize(var.type) * var.rows,
|
||||
VarElemPointer(tmp, c * var.rows));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1720,24 +1707,18 @@ void Debugger::WriteThroughPointer(const ShaderVariable &ptr, const ShaderVariab
|
||||
{
|
||||
// we can write a vector at a time if the matrix is column major
|
||||
apiWrapper->WriteBufferValue(bind, offset, VarTypeByteSize(var.type) * var.columns,
|
||||
var.value.uv);
|
||||
VarElemPointer(var, 0));
|
||||
}
|
||||
else
|
||||
{
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
if(VarTypeByteSize(var.type) == 8)
|
||||
apiWrapper->WriteBufferValue(bind, offset + c * matrixStride,
|
||||
VarTypeByteSize(var.type), &var.value.u64v[c]);
|
||||
else
|
||||
apiWrapper->WriteBufferValue(bind, offset + c * matrixStride,
|
||||
VarTypeByteSize(var.type), &var.value.uv[c]);
|
||||
}
|
||||
apiWrapper->WriteBufferValue(bind, offset + c * matrixStride, VarTypeByteSize(var.type),
|
||||
VarElemPointer(var, c));
|
||||
}
|
||||
}
|
||||
else if(type.type == DataType::ScalarType)
|
||||
{
|
||||
apiWrapper->WriteBufferValue(bind, offset, VarTypeByteSize(var.type), var.value.uv);
|
||||
apiWrapper->WriteBufferValue(bind, offset, VarTypeByteSize(var.type), VarElemPointer(var, 0));
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1770,21 +1751,13 @@ void Debugger::WriteThroughPointer(const ShaderVariable &ptr, const ShaderVariab
|
||||
{
|
||||
// two indices - selecting a scalar. scalar0 is the first index in the chain so it chooses
|
||||
// column
|
||||
if(VarTypeByteSize(storage->type) == 8)
|
||||
storage->value.u64v[scalar1 * storage->columns + scalar0] = val.value.u64v[0];
|
||||
else
|
||||
storage->value.uv[scalar1 * storage->columns + scalar0] = val.value.uv[0];
|
||||
copyComp(*storage, scalar1 * storage->columns + scalar0, val, 0);
|
||||
}
|
||||
else if(scalar0 != ~0U)
|
||||
{
|
||||
// one index, selecting a column
|
||||
for(uint32_t row = 0; row < storage->rows; row++)
|
||||
{
|
||||
if(VarTypeByteSize(storage->type) == 8)
|
||||
storage->value.u64v[row * storage->columns + scalar0] = val.value.u64v[row];
|
||||
else
|
||||
storage->value.uv[row * storage->columns + scalar0] = val.value.uv[row];
|
||||
}
|
||||
copyComp(*storage, row * storage->columns + scalar0, val, row);
|
||||
}
|
||||
}
|
||||
else
|
||||
@@ -1792,10 +1765,7 @@ void Debugger::WriteThroughPointer(const ShaderVariable &ptr, const ShaderVariab
|
||||
ClampScalars(apiWrapper, *storage, scalar0);
|
||||
|
||||
// vector case, selecting a scalar
|
||||
if(VarTypeByteSize(storage->type) == 8)
|
||||
storage->value.u64v[scalar0] = val.value.u64v[0];
|
||||
else
|
||||
storage->value.uv[scalar0] = val.value.uv[0];
|
||||
copyComp(*storage, scalar0, val, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -2057,8 +2027,8 @@ uint32_t Debugger::WalkVariable(
|
||||
|
||||
uint32_t childOffset = 0;
|
||||
|
||||
ShaderVariable len = GetActiveLane().ids[type.length];
|
||||
for(uint32_t i = 0; i < len.value.u.x; i++)
|
||||
uint32_t len = uintComp(GetActiveLane().ids[type.length], 0);
|
||||
for(uint32_t i = 0; i < len; i++)
|
||||
{
|
||||
if(outVar)
|
||||
{
|
||||
@@ -2101,11 +2071,167 @@ uint32_t Debugger::WalkVariable(
|
||||
return numLocations;
|
||||
}
|
||||
|
||||
template <typename FloatType>
|
||||
static void ApplyDerivative(uint32_t activeLaneIndex, uint32_t quadIndex, FloatType *dst,
|
||||
DebugAPIWrapper::DerivativeDeltas &derivs)
|
||||
{
|
||||
// We make the assumption that the coarse derivatives are generated from (0,0) in the quad, and
|
||||
// fine derivatives are generated from the destination index and its neighbours in X and Y.
|
||||
// This isn't spec'd but we must assume something and this will hopefully get us closest to
|
||||
// reproducing actual results.
|
||||
//
|
||||
// For debugging, we need members of the quad to be able to generate coarse and fine
|
||||
// derivatives.
|
||||
//
|
||||
// For (0,0) we only need the coarse derivatives to get our neighbours (1,0) and (0,1) which
|
||||
// will give us coarse and fine derivatives being identical.
|
||||
//
|
||||
// For the others we will need to use a combination of coarse and fine derivatives to get the
|
||||
// diagonal element in the quad. In the examples below, remember that the quad indices are:
|
||||
//
|
||||
// +---+---+
|
||||
// | 0 | 1 |
|
||||
// +---+---+
|
||||
// | 2 | 3 |
|
||||
// +---+---+
|
||||
//
|
||||
// And that we have definitions of the derivatives:
|
||||
//
|
||||
// ddx_coarse = (1,0) - (0,0)
|
||||
// ddy_coarse = (0,1) - (0,0)
|
||||
//
|
||||
// i.e. the same for all members of the quad
|
||||
//
|
||||
// ddx_fine = (x,y) - (1-x,y)
|
||||
// ddy_fine = (x,y) - (x,1-y)
|
||||
//
|
||||
// i.e. the difference to the neighbour of our desired invocation (the one we have the actual
|
||||
// inputs for, from gathering above).
|
||||
//
|
||||
// So e.g. if our thread is at (1,1) destIdx = 3
|
||||
//
|
||||
// (1,0) = (1,1) - ddx_fine
|
||||
// (0,1) = (1,1) - ddy_fine
|
||||
// (0,0) = (1,1) - ddy_fine - ddx_coarse
|
||||
//
|
||||
// and ddy_coarse is unused. For (1,0) destIdx = 1:
|
||||
//
|
||||
// (1,1) = (1,0) + ddy_fine
|
||||
// (0,1) = (1,0) - ddx_coarse + ddy_coarse
|
||||
// (0,0) = (1,0) - ddx_coarse
|
||||
//
|
||||
// and ddx_fine is unused (it's identical to ddx_coarse anyway)
|
||||
|
||||
// in the diagrams below * marks the active lane index.
|
||||
//
|
||||
// V and ^ == coarse ddy
|
||||
// , and ` == fine ddy
|
||||
// < and > == coarse ddx
|
||||
// { and } == fine ddx
|
||||
//
|
||||
// We are basically making one or two cardinal direction moves from the starting point
|
||||
// (activeLaneIndex) to the end point (quadIndex).
|
||||
RDCASSERTNOTEQUAL(activeLaneIndex, quadIndex);
|
||||
|
||||
#define ADD_DERIV(src) \
|
||||
for(int i = 0; i < 4; i++) \
|
||||
dst[i] += ((FloatType *)src.value.u64v)[i];
|
||||
#define SUB_DERIV(src) \
|
||||
for(int i = 0; i < 4; i++) \
|
||||
dst[i] -= ((FloatType *)src.value.u64v)[i];
|
||||
|
||||
switch(activeLaneIndex)
|
||||
{
|
||||
case 0:
|
||||
{
|
||||
// +---+---+
|
||||
// |*0 > 1 |
|
||||
// +-V-+-V-+
|
||||
// | 2 | 3 |
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0: break;
|
||||
case 1: ADD_DERIV(derivs.ddxcoarse); break;
|
||||
case 2: ADD_DERIV(derivs.ddycoarse); break;
|
||||
case 3:
|
||||
ADD_DERIV(derivs.ddxcoarse);
|
||||
ADD_DERIV(derivs.ddycoarse);
|
||||
break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 1:
|
||||
{
|
||||
// we need to use fine to get from 1 to 3 as coarse only ever involves 0->1 and 0->2
|
||||
// +---+---+
|
||||
// | 0 < 1*|
|
||||
// +-V-+-,-+
|
||||
// | 2 | 3 |
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0: SUB_DERIV(derivs.ddxcoarse); break;
|
||||
case 1: break;
|
||||
case 2:
|
||||
SUB_DERIV(derivs.ddxcoarse);
|
||||
ADD_DERIV(derivs.ddycoarse);
|
||||
break;
|
||||
case 3: ADD_DERIV(derivs.ddyfine); break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2:
|
||||
{
|
||||
// +---+---+
|
||||
// | 0 > 1 |
|
||||
// +-^-+---+
|
||||
// |*2 } 3 |
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0: SUB_DERIV(derivs.ddycoarse); break;
|
||||
case 1:
|
||||
SUB_DERIV(derivs.ddycoarse);
|
||||
ADD_DERIV(derivs.ddxcoarse);
|
||||
break;
|
||||
case 2: break;
|
||||
case 3: ADD_DERIV(derivs.ddxfine); break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 3:
|
||||
{
|
||||
// +---+---+
|
||||
// | 0 < 1 |
|
||||
// +---+-`-+
|
||||
// | 2 { 3*|
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0:
|
||||
SUB_DERIV(derivs.ddyfine);
|
||||
SUB_DERIV(derivs.ddxcoarse);
|
||||
break;
|
||||
case 1: SUB_DERIV(derivs.ddyfine); break;
|
||||
case 2: SUB_DERIV(derivs.ddxfine); break;
|
||||
case 3: break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
uint32_t Debugger::ApplyDerivatives(uint32_t quadIndex, const Decorations &curDecorations,
|
||||
uint32_t location, const DataType &inType, ShaderVariable &outVar)
|
||||
{
|
||||
// only floats have derivatives
|
||||
if(outVar.type == VarType::Float)
|
||||
if(outVar.type == VarType::Float || outVar.type == VarType::Half || outVar.type == VarType::Double)
|
||||
{
|
||||
ShaderBuiltin builtin = ShaderBuiltin::Undefined;
|
||||
if(curDecorations.flags & Decorations::HasBuiltIn)
|
||||
@@ -2121,157 +2247,19 @@ uint32_t Debugger::ApplyDerivatives(uint32_t quadIndex, const Decorations &curDe
|
||||
}
|
||||
}
|
||||
|
||||
// We make the assumption that the coarse derivatives are generated from (0,0) in the quad, and
|
||||
// fine derivatives are generated from the destination index and its neighbours in X and Y.
|
||||
// This isn't spec'd but we must assume something and this will hopefully get us closest to
|
||||
// reproducing actual results.
|
||||
//
|
||||
// For debugging, we need members of the quad to be able to generate coarse and fine
|
||||
// derivatives.
|
||||
//
|
||||
// For (0,0) we only need the coarse derivatives to get our neighbours (1,0) and (0,1) which
|
||||
// will give us coarse and fine derivatives being identical.
|
||||
//
|
||||
// For the others we will need to use a combination of coarse and fine derivatives to get the
|
||||
// diagonal element in the quad. In the examples below, remember that the quad indices are:
|
||||
//
|
||||
// +---+---+
|
||||
// | 0 | 1 |
|
||||
// +---+---+
|
||||
// | 2 | 3 |
|
||||
// +---+---+
|
||||
//
|
||||
// And that we have definitions of the derivatives:
|
||||
//
|
||||
// ddx_coarse = (1,0) - (0,0)
|
||||
// ddy_coarse = (0,1) - (0,0)
|
||||
//
|
||||
// i.e. the same for all members of the quad
|
||||
//
|
||||
// ddx_fine = (x,y) - (1-x,y)
|
||||
// ddy_fine = (x,y) - (x,1-y)
|
||||
//
|
||||
// i.e. the difference to the neighbour of our desired invocation (the one we have the actual
|
||||
// inputs for, from gathering above).
|
||||
//
|
||||
// So e.g. if our thread is at (1,1) destIdx = 3
|
||||
//
|
||||
// (1,0) = (1,1) - ddx_fine
|
||||
// (0,1) = (1,1) - ddy_fine
|
||||
// (0,0) = (1,1) - ddy_fine - ddx_coarse
|
||||
//
|
||||
// and ddy_coarse is unused. For (1,0) destIdx = 1:
|
||||
//
|
||||
// (1,1) = (1,0) + ddy_fine
|
||||
// (0,1) = (1,0) - ddx_coarse + ddy_coarse
|
||||
// (0,0) = (1,0) - ddx_coarse
|
||||
//
|
||||
// and ddx_fine is unused (it's identical to ddx_coarse anyway)
|
||||
|
||||
if(curDecorations.flags & Decorations::HasLocation)
|
||||
location = curDecorations.location;
|
||||
|
||||
DebugAPIWrapper::DerivativeDeltas derivs =
|
||||
apiWrapper->GetDerivative(builtin, location, component);
|
||||
apiWrapper->GetDerivative(builtin, location, component, outVar.type);
|
||||
|
||||
Vec4f &dst = *(Vec4f *)outVar.value.fv;
|
||||
|
||||
// in the diagrams below * marks the active lane index.
|
||||
//
|
||||
// V and ^ == coarse ddy
|
||||
// , and ` == fine ddy
|
||||
// < and > == coarse ddx
|
||||
// { and } == fine ddx
|
||||
//
|
||||
// We are basically making one or two cardinal direction moves from the starting point
|
||||
// (activeLaneIndex) to the end point (quadIndex).
|
||||
RDCASSERTNOTEQUAL(activeLaneIndex, quadIndex);
|
||||
|
||||
switch(activeLaneIndex)
|
||||
{
|
||||
case 0:
|
||||
{
|
||||
// +---+---+
|
||||
// |*0 > 1 |
|
||||
// +-V-+-V-+
|
||||
// | 2 | 3 |
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0: break;
|
||||
case 1: dst += derivs.ddxcoarse; break;
|
||||
case 2: dst += derivs.ddycoarse; break;
|
||||
case 3:
|
||||
dst += derivs.ddxcoarse;
|
||||
dst += derivs.ddycoarse;
|
||||
break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 1:
|
||||
{
|
||||
// we need to use fine to get from 1 to 3 as coarse only ever involves 0->1 and 0->2
|
||||
// +---+---+
|
||||
// | 0 < 1*|
|
||||
// +-V-+-,-+
|
||||
// | 2 | 3 |
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0: dst -= derivs.ddxcoarse; break;
|
||||
case 1: break;
|
||||
case 2:
|
||||
dst -= derivs.ddxcoarse;
|
||||
dst += derivs.ddycoarse;
|
||||
break;
|
||||
case 3: dst += derivs.ddyfine; break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 2:
|
||||
{
|
||||
// +---+---+
|
||||
// | 0 > 1 |
|
||||
// +-^-+---+
|
||||
// |*2 } 3 |
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0: dst -= derivs.ddycoarse; break;
|
||||
case 1:
|
||||
dst -= derivs.ddycoarse;
|
||||
dst += derivs.ddxcoarse;
|
||||
break;
|
||||
case 2: break;
|
||||
case 3: dst += derivs.ddxfine; break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case 3:
|
||||
{
|
||||
// +---+---+
|
||||
// | 0 < 1 |
|
||||
// +---+-`-+
|
||||
// | 2 { 3*|
|
||||
// +---+---+
|
||||
switch(quadIndex)
|
||||
{
|
||||
case 0:
|
||||
dst -= derivs.ddyfine;
|
||||
dst -= derivs.ddxcoarse;
|
||||
break;
|
||||
case 1: dst -= derivs.ddyfine; break;
|
||||
case 2: dst -= derivs.ddxfine; break;
|
||||
case 3: break;
|
||||
default: break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
if(outVar.type == VarType::Float)
|
||||
ApplyDerivative<float>(activeLaneIndex, quadIndex, outVar.value.fv, derivs);
|
||||
else if(outVar.type == VarType::Half)
|
||||
ApplyDerivative<half_float::half>(activeLaneIndex, quadIndex,
|
||||
(half_float::half *)outVar.value.u16v, derivs);
|
||||
else if(outVar.type == VarType::Double)
|
||||
ApplyDerivative<double>(activeLaneIndex, quadIndex, outVar.value.dv, derivs);
|
||||
}
|
||||
|
||||
// each row consumes a new location
|
||||
|
||||
@@ -26,6 +26,7 @@
|
||||
#include "driver/shaders/spirv/spirv_debug.h"
|
||||
#include "driver/shaders/spirv/spirv_editor.h"
|
||||
#include "driver/shaders/spirv/spirv_op_helpers.h"
|
||||
#include "driver/shaders/spirv/var_dispatch_helpers.h"
|
||||
#include "maths/formatpacking.h"
|
||||
#include "vk_core.h"
|
||||
#include "vk_debug.h"
|
||||
@@ -377,19 +378,22 @@ public:
|
||||
if(data.width == 0)
|
||||
return false;
|
||||
|
||||
if(coord.value.uv[0] > data.width || coord.value.uv[1] > data.height ||
|
||||
coord.value.uv[2] > data.depth)
|
||||
uint32_t coords[4];
|
||||
for(int i = 0; i < 4; i++)
|
||||
coords[i] = uintComp(coord, i);
|
||||
|
||||
if(coords[0] > data.width || coords[1] > data.height || coords[2] > data.depth)
|
||||
{
|
||||
m_pDriver->AddDebugMessage(
|
||||
MessageCategory::Execution, MessageSeverity::High, MessageSource::RuntimeWarning,
|
||||
StringFormat::Fmt(
|
||||
"Out of bounds access to image, coord %u,%u,%u outside of dimensions %ux%ux%u",
|
||||
coord.value.uv[0], coord.value.uv[1], coord.value.uv[2], data.width, data.height,
|
||||
data.depth));
|
||||
coords[0], coords[1], coords[2], data.width, data.height, data.depth));
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(output.value.uv, data.texel(coord.value.uv, sample), data.texelSize);
|
||||
RDCASSERTEQUAL(data.texelSize, VarTypeByteSize(output.type) * output.columns);
|
||||
memcpy(output.value.u64v, data.texel(coords, sample), data.texelSize);
|
||||
|
||||
return true;
|
||||
}
|
||||
@@ -402,19 +406,22 @@ public:
|
||||
if(data.width == 0)
|
||||
return false;
|
||||
|
||||
if(coord.value.uv[0] > data.width || coord.value.uv[1] > data.height ||
|
||||
coord.value.uv[2] > data.depth)
|
||||
uint32_t coords[4];
|
||||
for(int i = 0; i < 4; i++)
|
||||
coords[i] = uintComp(coord, i);
|
||||
|
||||
if(coords[0] > data.width || coords[1] > data.height || coords[2] > data.depth)
|
||||
{
|
||||
m_pDriver->AddDebugMessage(
|
||||
MessageCategory::Execution, MessageSeverity::High, MessageSource::RuntimeWarning,
|
||||
StringFormat::Fmt(
|
||||
"Out of bounds access to image, coord %u,%u,%u outside of dimensions %ux%ux%u",
|
||||
coord.value.uv[0], coord.value.uv[1], coord.value.uv[2], data.width, data.height,
|
||||
data.depth));
|
||||
coords[0], coords[1], coords[2], data.width, data.height, data.depth));
|
||||
return false;
|
||||
}
|
||||
|
||||
memcpy(data.texel(coord.value.uv, sample), value.value.uv, data.texelSize);
|
||||
RDCASSERTEQUAL(data.texelSize, VarTypeByteSize(value.type) * value.columns);
|
||||
memcpy(data.texel(coords, sample), value.value.u64v, data.texelSize);
|
||||
|
||||
return true;
|
||||
}
|
||||
@@ -440,28 +447,18 @@ public:
|
||||
const uint32_t typeSize = VarTypeByteSize(var.type);
|
||||
if(var.rows == 1)
|
||||
{
|
||||
if(component > 3)
|
||||
RDCERR("Unexpected component %u ", component);
|
||||
if(component + var.columns > 4)
|
||||
RDCERR("Unexpected component %u for column count %u", component, var.columns);
|
||||
|
||||
if(typeSize == 8)
|
||||
memcpy(var.value.u64v, &location_inputs[location].value.u64v[component],
|
||||
var.rows * var.columns * typeSize);
|
||||
else
|
||||
memcpy(var.value.uv, &location_inputs[location].value.uv[component],
|
||||
var.rows * var.columns * typeSize);
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
copyComp(var, c, location_inputs[location], component + c, var.type);
|
||||
}
|
||||
else
|
||||
{
|
||||
RDCASSERTEQUAL(component, 0);
|
||||
for(uint8_t r = 0; r < var.rows; r++)
|
||||
{
|
||||
for(uint8_t c = 0; c < var.columns; c++)
|
||||
{
|
||||
if(typeSize == 8)
|
||||
var.value.u64v[r * var.columns + c] = location_inputs[location + c].value.u64v[r];
|
||||
else
|
||||
var.value.uv[r * var.columns + c] = location_inputs[location + c].value.uv[r];
|
||||
}
|
||||
}
|
||||
copyComp(var, r * var.columns + c, location_inputs[location + c], r, var.type);
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -471,7 +468,7 @@ public:
|
||||
}
|
||||
|
||||
virtual DerivativeDeltas GetDerivative(ShaderBuiltin builtin, uint32_t location,
|
||||
uint32_t component) override
|
||||
uint32_t component, VarType type) override
|
||||
{
|
||||
if(builtin != ShaderBuiltin::Undefined)
|
||||
{
|
||||
@@ -495,10 +492,10 @@ public:
|
||||
|
||||
for(uint32_t src = component, dst = 0; src < 4; src++, dst++)
|
||||
{
|
||||
ret.ddxcoarse.fv[dst] = deriv.ddxcoarse.fv[src];
|
||||
ret.ddxfine.fv[dst] = deriv.ddxfine.fv[src];
|
||||
ret.ddycoarse.fv[dst] = deriv.ddycoarse.fv[src];
|
||||
ret.ddyfine.fv[dst] = deriv.ddyfine.fv[src];
|
||||
copyComp(ret.ddxcoarse, dst, deriv.ddxcoarse, src, type);
|
||||
copyComp(ret.ddxfine, dst, deriv.ddxfine, src, type);
|
||||
copyComp(ret.ddycoarse, dst, deriv.ddycoarse, src, type);
|
||||
copyComp(ret.ddyfine, dst, deriv.ddyfine, src, type);
|
||||
}
|
||||
|
||||
return ret;
|
||||
@@ -2841,7 +2838,8 @@ struct PSHit
|
||||
|
||||
static void CreatePSInputFetcher(rdcarray<uint32_t> &fragspv, uint32_t &structStride,
|
||||
VulkanCreationInfo::ShaderModuleReflection &shadRefl,
|
||||
StorageMode storageMode, bool usePrimitiveID, bool useSampleID)
|
||||
const uint32_t paramAlign, StorageMode storageMode,
|
||||
bool usePrimitiveID, bool useSampleID)
|
||||
{
|
||||
rdcspv::Editor editor(fragspv);
|
||||
|
||||
@@ -3769,7 +3767,7 @@ ShaderDebugTrace *VulkanReplay::DebugVertex(uint32_t eventId, uint32_t vertid, u
|
||||
if(Vulkan_Debug_ShaderDebugLogging())
|
||||
RDCLOG("Populating location %u", attr.location);
|
||||
|
||||
ShaderValue &val = locations[attr.location].value;
|
||||
ShaderVariable &var = locations[attr.location];
|
||||
|
||||
bytebuf data;
|
||||
|
||||
@@ -3839,10 +3837,7 @@ ShaderDebugTrace *VulkanReplay::DebugVertex(uint32_t eventId, uint32_t vertid, u
|
||||
"(index %u) in instance %u.",
|
||||
attr.location, attr.binding, vertid, idx, instid));
|
||||
|
||||
if(IsUIntFormat(attr.format) || IsSIntFormat(attr.format))
|
||||
val.u = {0, 0, 0, 1};
|
||||
else
|
||||
val.f = {0.0f, 0.0f, 0.0f, 1.0f};
|
||||
set0001(var);
|
||||
}
|
||||
else
|
||||
{
|
||||
@@ -3856,31 +3851,25 @@ ShaderDebugTrace *VulkanReplay::DebugVertex(uint32_t eventId, uint32_t vertid, u
|
||||
// this is the only packed UINT format
|
||||
Vec4u decoded = ConvertFromR10G10B10A2UInt(*(uint32_t *)data.data());
|
||||
|
||||
val.u.x = decoded.x;
|
||||
val.u.y = decoded.y;
|
||||
val.u.z = decoded.z;
|
||||
val.u.w = decoded.w;
|
||||
setUintComp(var, 0, decoded.x);
|
||||
setUintComp(var, 1, decoded.y);
|
||||
setUintComp(var, 2, decoded.z);
|
||||
setUintComp(var, 3, decoded.w);
|
||||
}
|
||||
else
|
||||
{
|
||||
for(uint32_t i = 0; i < fmt.compCount; i++)
|
||||
{
|
||||
const byte *src = data.data() + i * fmt.compByteWidth;
|
||||
if(fmt.compByteWidth == 8)
|
||||
memcpy(&val.u64v[i], src, fmt.compByteWidth);
|
||||
else
|
||||
memcpy(&val.uv[i], src, fmt.compByteWidth);
|
||||
}
|
||||
RDCASSERTEQUAL(fmt.compByteWidth, VarTypeByteSize(var.type));
|
||||
memcpy(var.value.u64v, data.data(), fmt.compByteWidth * fmt.compCount);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
FloatVector decoded = DecodeFormattedComponents(fmt, data.data());
|
||||
|
||||
val.f.x = decoded.x;
|
||||
val.f.y = decoded.y;
|
||||
val.f.z = decoded.z;
|
||||
val.f.w = decoded.w;
|
||||
setFloatComp(var, 0, decoded.x);
|
||||
setFloatComp(var, 1, decoded.y);
|
||||
setFloatComp(var, 2, decoded.z);
|
||||
setFloatComp(var, 3, decoded.w);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -4053,8 +4042,17 @@ ShaderDebugTrace *VulkanReplay::DebugPixel(uint32_t eventId, uint32_t x, uint32_
|
||||
if(!Vulkan_Debug_PSDebugDumpDirPath().empty())
|
||||
FileIO::WriteAll(Vulkan_Debug_PSDebugDumpDirPath() + "/debug_psinput_before.spv", fragspv);
|
||||
|
||||
uint32_t paramAlign = 16;
|
||||
|
||||
for(const SigParameter &sig : shadRefl.refl.inputSignature)
|
||||
{
|
||||
if(VarTypeByteSize(sig.varType) * sig.compCount > paramAlign)
|
||||
paramAlign = 32;
|
||||
}
|
||||
|
||||
uint32_t structStride = 0;
|
||||
CreatePSInputFetcher(fragspv, structStride, shadRefl, storageMode, usePrimitiveID, useSampleID);
|
||||
CreatePSInputFetcher(fragspv, structStride, shadRefl, paramAlign, storageMode, usePrimitiveID,
|
||||
useSampleID);
|
||||
|
||||
if(!Vulkan_Debug_PSDebugDumpDirPath().empty())
|
||||
FileIO::WriteAll(Vulkan_Debug_PSDebugDumpDirPath() + "/debug_psinput_after.spv", fragspv);
|
||||
@@ -4486,14 +4484,15 @@ ShaderDebugTrace *VulkanReplay::DebugPixel(uint32_t eventId, uint32_t x, uint32_
|
||||
rdcspv::Debugger *debugger = new rdcspv::Debugger;
|
||||
debugger->Parse(shader.spirv.GetSPIRV());
|
||||
|
||||
// the data immediately follows the PSHit header. Every piece of data is vec4 aligned, and the
|
||||
// output is in input signature order.
|
||||
// the data immediately follows the PSHit header. Every piece of data is uniformly aligned,
|
||||
// either 16-byte by default or 32-byte if larger components exist. The output is in input
|
||||
// signature order.
|
||||
byte *PSInputs = (byte *)(winner + 1);
|
||||
Vec4f *value = (Vec4f *)(PSInputs + 0 * structStride);
|
||||
Vec4f *ddxcoarse = (Vec4f *)(PSInputs + 1 * structStride);
|
||||
Vec4f *ddycoarse = (Vec4f *)(PSInputs + 2 * structStride);
|
||||
Vec4f *ddxfine = (Vec4f *)(PSInputs + 3 * structStride);
|
||||
Vec4f *ddyfine = (Vec4f *)(PSInputs + 4 * structStride);
|
||||
byte *value = (byte *)(PSInputs + 0 * structStride);
|
||||
byte *ddxcoarse = (byte *)(PSInputs + 1 * structStride);
|
||||
byte *ddycoarse = (byte *)(PSInputs + 2 * structStride);
|
||||
byte *ddxfine = (byte *)(PSInputs + 3 * structStride);
|
||||
byte *ddyfine = (byte *)(PSInputs + 4 * structStride);
|
||||
|
||||
for(size_t i = 0; i < shadRefl.refl.inputSignature.size(); i++)
|
||||
{
|
||||
@@ -4515,15 +4514,16 @@ ShaderDebugTrace *VulkanReplay::DebugPixel(uint32_t eventId, uint32_t x, uint32_
|
||||
builtin ? apiWrapper->builtin_derivatives[param.systemValue]
|
||||
: apiWrapper->location_derivatives[param.regIndex];
|
||||
|
||||
uint32_t comp = Bits::CountTrailingZeroes(uint32_t(param.regChannelMask));
|
||||
const uint32_t comp = Bits::CountTrailingZeroes(uint32_t(param.regChannelMask));
|
||||
const uint32_t elemSize = VarTypeByteSize(param.varType);
|
||||
|
||||
const size_t sz = sizeof(Vec4f) - sizeof(uint32_t) * comp;
|
||||
const size_t sz = elemSize * param.compCount;
|
||||
|
||||
memcpy(&var.value.uv[comp], &value[i], sz);
|
||||
memcpy(&deriv.ddxcoarse.fv[comp], &ddxcoarse[i], sz);
|
||||
memcpy(&deriv.ddycoarse.fv[comp], &ddycoarse[i], sz);
|
||||
memcpy(&deriv.ddxfine.fv[comp], &ddxfine[i], sz);
|
||||
memcpy(&deriv.ddyfine.fv[comp], &ddyfine[i], sz);
|
||||
memcpy(((byte *)var.value.u64v) + elemSize * comp, value + i * paramAlign, sz);
|
||||
memcpy(((byte *)deriv.ddxcoarse.value.u64v) + elemSize * comp, ddxcoarse + i * paramAlign, sz);
|
||||
memcpy(((byte *)deriv.ddycoarse.value.u64v) + elemSize * comp, ddycoarse + i * paramAlign, sz);
|
||||
memcpy(((byte *)deriv.ddxfine.value.u64v) + elemSize * comp, ddxfine + i * paramAlign, sz);
|
||||
memcpy(((byte *)deriv.ddyfine.value.u64v) + elemSize * comp, ddyfine + i * paramAlign, sz);
|
||||
}
|
||||
|
||||
ret = debugger->BeginDebug(apiWrapper, ShaderStage::Pixel, entryPoint, spec,
|
||||
|
||||
Reference in New Issue
Block a user