From a8e7a32569f88087442cdacdd98e4304b4a633f9 Mon Sep 17 00:00:00 2001 From: Jack Elliott Date: Fri, 11 Sep 2026 05:11:15 +1200 Subject: [PATCH] [HLSL] Test bounded vector accumulation and exact I32 contention Add an opt-in bounded raw UAV descriptor table to the existing vector accumulation runner. Preserve its fully initialised root-SRV source and all existing root-UAV cases. Check actual 64-byte destination alignment. Exercise F16 and F32 with a view containing the complete vector but not its guards, a view containing four of eight elements, and a view ending at the vector start. Independently specified seeded results distinguish accumulation from store, missing in-view updates, widened bounds and outside writes. Only a partially viewed input vector admits either the per-element result or a whole-operation no-op; the larger guarded destination does not itself make the vector OOB. Typed guards belong to the exact element comparison, while the byte oracle checks the prologue. Add signed I32 contention using native int32_t data and the existing four-component base-four dispatch-ID pattern. Across 256 invocations, each digit contributes 384. Base {3,-7,11,-13} and seed {101,-203,307,-409} yield {1253,-1611,3507,-3353}; all partial sums are in range. Every invocation contributes a distinct whole vector, so one unequal-vector drop/replay pair cannot cancel component-wise. This does not claim detection of arbitrary cancelling combinations or rely on a scalar-total check. Fresh baseline and candidate use identical compiler/validator binaries built from ef47f5aa. All 91 existing per-test outcomes are unchanged. Both bounded vector tests pass on preview WARP. The new I32 case legitimately skips: SInt32 reports UAV=0. Its exact shader compiles and validates as v4i32, but supported hardware and runtime-oracle non-vacuity remain unproved. Existing baseline failures are not changed. Local feature checkpoint before negative controls. This AI-assisted engineering rationale and implementation await human review; publication is not authorised. Assisted-by: GitHub Copilot Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 83725f5d-8e98-4c1d-91ee-ad47629e007b --- .../clang/unittests/HLSLExec/LinAlgTests.cpp | 142 +++++++++++++++++- 1 file changed, 138 insertions(+), 4 deletions(-) diff --git a/tools/clang/unittests/HLSLExec/LinAlgTests.cpp b/tools/clang/unittests/HLSLExec/LinAlgTests.cpp index 2ea0a59952..9937c7ceff 100644 --- a/tools/clang/unittests/HLSLExec/LinAlgTests.cpp +++ b/tools/clang/unittests/HLSLExec/LinAlgTests.cpp @@ -3693,8 +3693,11 @@ class DxilConf_SM610_LinAlg { TEST_METHOD(VectorAccumulateDescriptor_Thread_F16); TEST_METHOD(VectorAccumulateDescriptor_Thread_F16_Length8_NonZero); TEST_METHOD(VectorAccumulateDescriptor_Thread_F32_Length8_NonZero); + TEST_METHOD(VectorAccumulateDescriptorOOB_Thread_F16); + TEST_METHOD(VectorAccumulateDescriptorOOB_Thread_F32); TEST_METHOD(VectorAccumulateDescriptorContention_Thread_F16); TEST_METHOD(VectorAccumulateDescriptorContention_Thread_F32_OrderInvariant); + TEST_METHOD(VectorAccumulateDescriptorContention_Thread_I32); private: CComPtr D3DDevice; @@ -9241,7 +9244,7 @@ static void runVectorAccumulateDescriptor( const cpu_oracle::TypedMatrix &Initial, const cpu_oracle::TypedMatrix &Expected, UINT StartOffsetBytes, std::wstring PublicRule, bool Verbose, UINT NumThreads = 1, - UINT DispatchX = 1) { + UINT DispatchX = 1, std::optional OutputViewBytes = std::nullopt) { VERIFY_ARE_EQUAL(1u, Input.M, "Vector input must have one row"); VERIFY_ARE_EQUAL(Input.compType(), Initial.compType(), "Input and destination component types must match"); @@ -9254,6 +9257,8 @@ static void runVectorAccumulateDescriptor( VERIFY_IS_GREATER_THAN_OR_EQUAL( Initial.totalElements(), Input.totalElements(), "Destination must hold the input vector and any guard elements"); + VERIFY_IS_TRUE(StartOffsetBytes % 64 == 0, + "Vector start offset must preserve 64-byte alignment"); const cpu_oracle::MatrixBufferLayout InputLayout = { MatrixLayout::RowMajor, @@ -9277,6 +9282,15 @@ static void runVectorAccumulateDescriptor( if (!OutputSize) return; + if (OutputViewBytes) { + VERIFY_IS_TRUE(*OutputViewBytes > 0 && *OutputViewBytes <= *OutputSize, + "The bounded UAV view must fit the destination buffer"); + hlsl_test::LogCommentFmt( + L"Vector accumulation bounded UAV: view=%zu bytes, vector offset=%u, " + L"vector size=%zu, destination size=%zu", + *OutputViewBytes, StartOffsetBytes, *InputSize, *OutputSize); + } + std::vector InputBytes(*InputSize); std::vector InitialBytes(*OutputSize); // Seed the destination with poison so the bytes the accumulation must leave @@ -9304,12 +9318,20 @@ static void runVectorAccumulateDescriptor( compileShader(DxcSupport, VectorAccumulateDescriptorShader, "cs_6_10", Args, Verbose); + const char *RootSignature = OutputViewBytes + ? "SRV(t0), DescriptorTable(UAV(u1))" + : "SRV(t0), UAV(u1)"; auto Op = createComputeOp(VectorAccumulateDescriptorShader, "cs_6_10", - "SRV(t0), UAV(u1)", Args.c_str(), DispatchX); + RootSignature, Args.c_str(), DispatchX); addSRVBuffer(Op.get(), "Input", InputBytes.size(), "byname"); addUAVBuffer(Op.get(), "Output", InitialBytes.size(), true, "byname"); addRootView(Op.get(), 0, "Input"); - addRootView(Op.get(), 1, "Output"); + if (OutputViewBytes) { + addHeapRawUAV(Op.get(), "ResHeap", "Output", *OutputViewBytes); + addRootTable(Op.get(), 1, "ResHeap"); + } else { + addRootView(Op.get(), 1, "Output"); + } auto Result = runShaderOp( Device, DxcSupport, std::move(Op), @@ -9327,12 +9349,28 @@ static void runVectorAccumulateDescriptor( "Vector accumulation initializer size mismatch"); if (Source->size() == Data.size()) std::memcpy(Data.data(), Source->data(), Data.size()); + }, + /*PostDispatchCallback=*/nullptr, + [StartOffsetBytes](ID3D12GraphicsCommandList *, st::ShaderOpTest *Test) { + ID3D12Resource *Output = nullptr; + Test->GetResource("Output", &Output); + VERIFY_IS_NOT_NULL(Output); + VERIFY_IS_TRUE( + (Output->GetGPUVirtualAddress() + StartOffsetBytes) % 64 == 0, + "Vector destination must meet its declared 64-byte alignment"); }); MappedData OutData; Result->Test->GetReadBackData("Output", &OutData); + // Bounds apply to the input vector, not the larger guarded destination. + const bool PartiallyInView = OutputViewBytes && + *OutputViewBytes > StartOffsetBytes && + *OutputViewBytes - StartOffsetBytes < *InputSize; const cpu_oracle::MatrixResultOracle Oracle = - cpu_oracle::exactResult(Expected, std::move(PublicRule)); + PartiallyInView + ? cpu_oracle::permittedResults({Expected, Initial}, + std::move(PublicRule)) + : cpu_oracle::exactResult(Expected, std::move(PublicRule)); VERIFY_IS_TRUE(cpu_oracle::verifyMatrixBuffer(OutData.data(), OutData.size(), OutputLayout, Oracle, Verbose)); VERIFY_IS_TRUE(cpu_oracle::verifyUntouchedBytes( @@ -9450,6 +9488,73 @@ void DxilConf_SM610_LinAlg:: VerboseLogging); } +template +static void +runVectorAccumulateDescriptorOutOfBounds(ID3D12Device *Device, + dxc::SpecificDllLoader &DxcSupport, + LPCWSTR CaseName, bool Verbose) { + if (!accumulateStoreApplicable( + Device, cpu_oracle::ComponentTraits::CompType, + linalg_test::AtomicDestination::RWByteAddressBuffer, CaseName)) + return; + + const auto Input = + cpu_oracle::makeTypedMatrix(1, 8, + {T(-3.0f), T(2.0f), T(5.0f), T(-1.0f), + T(4.0f), T(1.0f), T(-2.0f), T(6.0f)}); + const auto Initial = cpu_oracle::makeTypedMatrix( + 1, 10, + {T(10.0f), T(11.0f), T(12.0f), T(13.0f), T(14.0f), T(15.0f), T(16.0f), + T(17.0f), T(123.0f), T(-321.0f)}); + const auto Accumulated = cpu_oracle::makeTypedMatrix( + 1, 10, + {T(7.0f), T(13.0f), T(17.0f), T(12.0f), T(18.0f), T(16.0f), T(14.0f), + T(23.0f), T(123.0f), T(-321.0f)}); + const auto PartiallyAccumulated = cpu_oracle::makeTypedMatrix( + 1, 10, + {T(7.0f), T(13.0f), T(17.0f), T(12.0f), T(14.0f), T(15.0f), T(16.0f), + T(17.0f), T(123.0f), T(-321.0f)}); + VERIFY_IS_TRUE(Input.has_value() && Initial.has_value() && + Accumulated.has_value() && + PartiallyAccumulated.has_value(), + "Unable to construct vector descriptor bounds fixtures"); + if (!Input || !Initial || !Accumulated || !PartiallyAccumulated) + return; + + constexpr UINT StartOffsetBytes = 64; + const size_t ElementBytes = elementSize(Input->compType()); + // The full vector fits this view even though its trailing guards do not. + runVectorAccumulateDescriptor( + Device, DxcSupport, *Input, *Initial, *Accumulated, StartOffsetBytes, + L"A fully in-view vector must accumulate onto its non-zero seed", Verbose, + /*NumThreads=*/1, /*DispatchX=*/1, + /*OutputViewBytes=*/StartOffsetBytes + 8 * ElementBytes); + runVectorAccumulateDescriptor( + Device, DxcSupport, *Input, *Initial, *PartiallyAccumulated, + StartOffsetBytes, + L"Proposal 0035 permits either whole-operation or per-element no-op " + L"for vector accumulation crossing the descriptor bound", + Verbose, /*NumThreads=*/1, /*DispatchX=*/1, + /*OutputViewBytes=*/StartOffsetBytes + 4 * ElementBytes); + runVectorAccumulateDescriptor( + Device, DxcSupport, *Input, *Initial, *Initial, StartOffsetBytes, + L"A wholly out-of-view vector must leave the destination unchanged", + Verbose, /*NumThreads=*/1, /*DispatchX=*/1, + /*OutputViewBytes=*/StartOffsetBytes); +} + +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorOOB_Thread_F16() { + runVectorAccumulateDescriptorOutOfBounds( + D3DDevice, DxcSupport, L"VectorAccumulateDescriptorOOB_Thread_F16", + VerboseLogging); +} + +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorOOB_Thread_F32() { + runVectorAccumulateDescriptorOutOfBounds( + D3DDevice, DxcSupport, L"VectorAccumulateDescriptorOOB_Thread_F32", + VerboseLogging); +} + // The single-threaded cases above show that an accumulation lands, not that // concurrent accumulations all land. These dispatch many threads across many // groups at one destination. Each component carries one base-four digit of the @@ -9552,6 +9657,35 @@ void DxilConf_SM610_LinAlg:: VerboseLogging, VectorContentionThreads, VectorContentionGroups); } +void DxilConf_SM610_LinAlg::VectorAccumulateDescriptorContention_Thread_I32() { + if (!accumulateStoreApplicable( + D3DDevice, ComponentType::I32, + linalg_test::AtomicDestination::RWByteAddressBuffer, + L"VectorAccumulateDescriptorContention_Thread_I32")) + return; + + const auto Input = + cpu_oracle::makeTypedMatrix(1, 4, {3, -7, 11, -13}); + const auto Initial = cpu_oracle::makeTypedMatrix( + 1, 6, {101, -203, 307, -409, 123456789, -987654321}); + // Each base-four digit contributes 64 * (0 + 1 + 2 + 3) = 384. + // Every partial sum lies between the initial value and the exact result. + const auto Expected = cpu_oracle::makeTypedMatrix( + 1, 6, {1253, -1611, 3507, -3353, 123456789, -987654321}); + VERIFY_IS_TRUE(Input.has_value()); + VERIFY_IS_TRUE(Initial.has_value()); + VERIFY_IS_TRUE(Expected.has_value()); + if (!Input || !Initial || !Expected) + return; + + runVectorAccumulateDescriptor( + D3DDevice, DxcSupport, *Input, *Initial, *Expected, + /*StartOffsetBytes=*/64, + L"Exact signed I32 vector accumulation from 256 distinct contending " + L"invocations, with untouched trailing guards", + VerboseLogging, VectorContentionThreads, VectorContentionGroups); +} + void DxilConf_SM610_LinAlg::MatVecMul_Thread_4x8_F16_NonUniform() { const matvec_interpretation::CaseData Case = matvec_interpretation::makeNonUniformF16Case(MatrixLayout::RowMajor);