From e7e5531f66ca4613189e8a13f76bc167a75e5661 Mon Sep 17 00:00:00 2001 From: Bram Stolk Date: Thu, 10 Sep 2026 22:29:48 -0700 Subject: [PATCH 1/2] fix: detect access past end of peer block to avoid gpu crash We cannot assume that the peer's blocks are contiguous. This fixes a real gpu crash, triggered in real-world use. The command will now fail cpu-side instead. A follow-up commit will address avoiding this case. The error now surfaces as ZE_RESULT_ERROR_OUT_OF_DEVICE_MEMORY. But an accurate zeDriverGetLastErrorDescription is set. The guard covers all fourteen resolveAlignedAllocation callers: appendMemoryCopyRegion appendMemoryFill appendImageCopy* appendSignalEvent appendWaitOnMemory appendWriteToMemory Related-To: GSD-13429 Signed-off-by: Bram Stolk --- level_zero/core/source/cmdlist/cmdlist_hw.inl | 21 ++++++ .../sources/cmdlist/test_cmdlist_6.cpp | 66 +++++++++++++++++++ 2 files changed, 87 insertions(+) diff --git a/level_zero/core/source/cmdlist/cmdlist_hw.inl b/level_zero/core/source/cmdlist/cmdlist_hw.inl index d5fa9ffd561a8..01622d63c736f 100644 --- a/level_zero/core/source/cmdlist/cmdlist_hw.inl +++ b/level_zero/core/source/cmdlist/cmdlist_hw.inl @@ -3500,6 +3500,27 @@ inline AlignedAllocationData CommandListCoreFamily::resolveAligne } if (svmAllocFound) { + // Blocks of a virtual reservation are imported on a peer device one at a time, + // each at an address of its own, so the reservation is not contiguous there. A + // range that crosses a block end would address memory the peer never mapped, so + // reject it instead of letting the copy engine walk into it and hang. + auto *ownerAlloc = svmAlloc->gpuAllocations.getDefaultGraphicsAllocation(); + if (ownerAlloc != nullptr && bufferSize > 0u && svmAlloc->virtualReservationData != nullptr && + device->getDriverHandle()->isRemoteResourceNeeded(*svmAlloc, device)) { + const uint64_t blockEnd = ownerAlloc->getGpuAddress() + svmAlloc->size; + const uint64_t rangeEnd = castToUint64(ptr) + bufferSize; + if (rangeEnd > blockEnd) { + CREATE_DEBUG_STRING(str, "Peer access at 0x%llx runs %llu bytes past the end of its virtual reservation block\n", + static_cast(castToUint64(ptr)), + static_cast(rangeEnd - blockEnd)); + device->getDriverHandle()->setErrorDescription(std::string(str.get())); + PRINT_STRING(NEO::debugManager.flags.PrintDebugMessages.get(), stderr, + "Peer access at 0x%llx runs %llu bytes past the end of its virtual reservation block\n", + static_cast(castToUint64(ptr)), + static_cast(rangeEnd - blockEnd)); + return AlignedAllocationData::invalid(); + } + } return alignSvmAllocationData(device, svmAlloc, buffer, sourcePtr, sshAlignmentOffset); } diff --git a/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp b/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp index 26bf9650f37e3..873c84e6fa00c 100644 --- a/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp +++ b/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp @@ -1835,6 +1835,72 @@ HWTEST_F(CommandListTest, givenComputeCommandListWhenMemoryCopyWithReservedDevic EXPECT_EQ(ZE_RESULT_SUCCESS, res); } +using MultiDeviceReservationCommandListTest = Test; + +HWTEST_F(MultiDeviceReservationCommandListTest, givenPeerAccessSpanningReservationBlocksWhenAppendingMemoryCopyThenCopyIsRejectedInsteadOfHanging) { + auto deviceOwner = driverHandle->devices[0]; + auto devicePeer = driverHandle->devices[1]; + + auto commandList = std::make_unique>>(); + commandList->initialize(devicePeer, NEO::EngineGroupType::renderCompute, 0u); + + void *buffer = nullptr; + size_t size = MemoryConstants::pageSize64k; + size_t reservationSize = size * 2; + + ASSERT_EQ(ZE_RESULT_SUCCESS, context->reserveVirtualMem(nullptr, reservationSize, &buffer)); + ze_physical_mem_desc_t desc = {}; + desc.size = size; + ze_physical_mem_handle_t phys0, phys1; + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(deviceOwner->toHandle(), &desc, &phys0)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(deviceOwner->toHandle(), &desc, &phys1)); + void *secondBlock = reinterpret_cast(reinterpret_cast(buffer) + size); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(buffer, size, phys0, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(secondBlock, size, phys1, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + + // the peer maps each block separately, so a range crossing the block end is refused + auto spanning = commandList->resolveAlignedAllocation(devicePeer, buffer, reservationSize, nullptr, {}); + EXPECT_EQ(nullptr, spanning.alloc); + + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(buffer, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(secondBlock, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->freeVirtualMem(buffer, reservationSize)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys0)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys1)); +} + +HWTEST_F(CommandListTest, givenReservationOnSameDeviceSpanningBlocksWhenResolvingAlignedAllocationThenCopyIsNotRejected) { + auto commandList = std::make_unique>>(); + commandList->initialize(device, NEO::EngineGroupType::renderCompute, 0u); + + driverHandle->devices[0]->getNEODevice()->getExecutionEnvironment()->rootDeviceEnvironments[0]->memoryOperationsInterface = + std::make_unique(); + + void *buffer = nullptr; + size_t size = MemoryConstants::pageSize64k; + size_t reservationSize = size * 2; + + ASSERT_EQ(ZE_RESULT_SUCCESS, context->reserveVirtualMem(nullptr, reservationSize, &buffer)); + ze_physical_mem_desc_t desc = {}; + desc.size = size; + ze_physical_mem_handle_t phys0, phys1; + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(device->toHandle(), &desc, &phys0)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(device->toHandle(), &desc, &phys1)); + void *secondBlock = reinterpret_cast(reinterpret_cast(buffer) + size); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(buffer, size, phys0, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(secondBlock, size, phys1, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + + // no peer access is involved, the reservation is contiguous here and stays usable + auto spanning = commandList->resolveAlignedAllocation(device, buffer, reservationSize, nullptr, {}); + EXPECT_NE(nullptr, spanning.alloc); + + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(buffer, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(secondBlock, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->freeVirtualMem(buffer, reservationSize)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys0)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys1)); +} + HWTEST_F(CommandListTest, givenReservedDeviceAllocationWhenResolvingAlignedAllocationThenVirtualMemoryReservationMapLockIsTaken) { auto commandList = std::make_unique>>(); commandList->initialize(device, NEO::EngineGroupType::renderCompute, 0u); From 2fe14a584642c513d5cc8ab54b7d5a79ce97ac79 Mon Sep 17 00:00:00 2001 From: Bram Stolk Date: Thu, 10 Sep 2026 22:48:15 -0700 Subject: [PATCH 2/2] fix: split peer copies before reaching end of block Memory blocks on the peer are possibly not contiguous. This means that an integral copy may not be possible. Instead, split the copies at block-ends. This adds a helper function to find the splits. Verified as correct on real hardware (2x Intel Arc Pro B70) which crashed before, but now correctly copies. Resolves: GSD-13429 Signed-off-by: Bram Stolk --- level_zero/core/source/cmdlist/cmdlist_hw.h | 1 + level_zero/core/source/cmdlist/cmdlist_hw.inl | 55 +++++++++++++ .../core/test/unit_tests/mocks/mock_cmdlist.h | 1 + .../sources/cmdlist/test_cmdlist_6.cpp | 81 +++++++++++++++++++ 4 files changed, 138 insertions(+) diff --git a/level_zero/core/source/cmdlist/cmdlist_hw.h b/level_zero/core/source/cmdlist/cmdlist_hw.h index 3edafe25f1b35..b23e05fd74be7 100644 --- a/level_zero/core/source/cmdlist/cmdlist_hw.h +++ b/level_zero/core/source/cmdlist/cmdlist_hw.h @@ -439,6 +439,7 @@ struct CommandListCoreFamily : public CommandList { uint64_t getInputBufferSize(NEO::ImageType imageType, uint32_t bufferRowPitch, uint64_t bufferSlicePitch, const ze_image_region_t *region, size_t pixelSize); MOCKABLE_VIRTUAL AlignedAllocationData resolveAlignedAllocation(Device *device, const void *buffer, uint64_t bufferSize, const MemAllocInfo *bufferAllocInfo, const ResolveAlignedAllocationFlags &flags); AlignedAllocationData alignSvmAllocationData(Device *device, NEO::SvmAllocationData *svmAlloc, const void *buffer, uintptr_t sourcePtr, size_t sshAlignmentOffset); + size_t bytesToPeerReservationBlockEnd(const void *ptr, size_t remaining); AlignedAllocationData alignImportedHostAllocationData(NEO::GraphicsAllocation *importedHostAlloc, void *ptr); AlignedAllocationData alignExplicitAllocationData(NEO::GraphicsAllocation *alloc, void *ptr); AlignedAllocationData alignCachedHostAllocationData(NEO::GraphicsAllocation *cachedHostAlloc, uintptr_t sourcePtr, size_t sshAlignmentOffset); diff --git a/level_zero/core/source/cmdlist/cmdlist_hw.inl b/level_zero/core/source/cmdlist/cmdlist_hw.inl index 01622d63c736f..d6bece426cefa 100644 --- a/level_zero/core/source/cmdlist/cmdlist_hw.inl +++ b/level_zero/core/source/cmdlist/cmdlist_hw.inl @@ -2316,6 +2316,35 @@ void CommandListCoreFamily::addHostFunctionToPatchCommands(const } } +// Blocks of a virtual reservation are imported on a peer device one at a time, each +// at a virtual address the peer heap picks, so they are not contiguous there. Return +// how many of the remaining bytes stay inside the block that holds ptr, so a peer copy +// can be split on that boundary. Returns remaining when ptr needs no peer access. +template +size_t CommandListCoreFamily::bytesToPeerReservationBlockEnd(const void *ptr, size_t remaining) { + NEO::SvmAllocationData *allocData = nullptr; + auto *driverHandle = this->device->getDriverHandle(); + if (!driverHandle->findAllocationDataForRange(const_cast(ptr), 1u, allocData) || allocData == nullptr) { + return remaining; + } + if (allocData->virtualReservationData == nullptr) { + return remaining; + } + auto *ownerAlloc = allocData->gpuAllocations.getDefaultGraphicsAllocation(); + if (ownerAlloc == nullptr) { + return remaining; + } + if (!driverHandle->isRemoteResourceNeeded(*allocData, this->device)) { + return remaining; + } + const uint64_t blockEnd = ownerAlloc->getGpuAddress() + allocData->size; + const uint64_t current = castToUint64(const_cast(ptr)); + if (current >= blockEnd) { + return remaining; + } + return std::min(remaining, static_cast(blockEnd - current)); +} + template ze_result_t CommandListCoreFamily::appendRecordedBcsSplit(void *dstptr, const void *srcptr, size_t size, ze_event_handle_t hSignalEvent, uint32_t numWaitEvents, ze_event_handle_t *phWaitEvents, CmdListMemoryCopyParams &memoryCopyParams) { @@ -2365,6 +2394,32 @@ ze_result_t CommandListCoreFamily::appendMemoryCopy(void *dstptr, return appendRecordedBcsSplit(dstptr, srcptr, size, hSignalEvent, numWaitEvents, phWaitEvents, memoryCopyParams); } + if (!memoryCopyParams.bscSplitEnabled && size > 0u) { + const size_t firstChunk = std::min(bytesToPeerReservationBlockEnd(dstptr, size), + bytesToPeerReservationBlockEnd(srcptr, size)); + if (firstChunk > 0u && firstChunk < size) { + size_t done = 0u; + while (done < size) { + const size_t left = size - done; + void *subDst = ptrOffset(dstptr, done); + const void *subSrc = ptrOffset(srcptr, done); + const size_t chunk = std::min(bytesToPeerReservationBlockEnd(subDst, left), + bytesToPeerReservationBlockEnd(subSrc, left)); + const bool isLast = (done + chunk >= size); + auto ret = appendMemoryCopy(subDst, subSrc, chunk, + isLast ? hSignalEvent : nullptr, + done == 0u ? numWaitEvents : 0u, + done == 0u ? phWaitEvents : nullptr, + memoryCopyParams, nullptr, nullptr); + if (ret != ZE_RESULT_SUCCESS) { + return ret; + } + done += chunk; + } + return ZE_RESULT_SUCCESS; + } + } + auto allocSize = NEO::getIfValid(memoryCopyParams.bcsSplitTotalDstSize, size); auto dstAllocationStruct = resolveAlignedAllocation(this->device, NEO::getIfValid(memoryCopyParams.bcsSplitBaseDstPtr, dstptr), allocSize, dstAllocInfo, {.sharedSystemEnabled = sharedSystemEnabled, .copyOffload = isCopyOffloadEnabled()}); auto srcAllocationStruct = resolveAlignedAllocation(this->device, NEO::getIfValid(memoryCopyParams.bcsSplitBaseSrcPtr, srcptr), allocSize, srcAllocInfo, {.sharedSystemEnabled = sharedSystemEnabled, .hostCopyAllowed = true, .copyOffload = isCopyOffloadEnabled()}); diff --git a/level_zero/core/test/unit_tests/mocks/mock_cmdlist.h b/level_zero/core/test/unit_tests/mocks/mock_cmdlist.h index 39828b3159bcd..f365fd0187104 100644 --- a/level_zero/core/test/unit_tests/mocks/mock_cmdlist.h +++ b/level_zero/core/test/unit_tests/mocks/mock_cmdlist.h @@ -61,6 +61,7 @@ struct WhiteBox<::L0::CommandListCoreFamily> using BaseClass::applyMemoryRangesBarrier; using BaseClass::arePostBlitWACmdsRequired; using BaseClass::bcsSplitMode; + using BaseClass::bytesToPeerReservationBlockEnd; using BaseClass::clearCommandsToPatch; using BaseClass::closedCmdList; using BaseClass::cmdListHeapAddressModel; diff --git a/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp b/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp index 873c84e6fa00c..47806815f0beb 100644 --- a/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp +++ b/level_zero/core/test/unit_tests/sources/cmdlist/test_cmdlist_6.cpp @@ -1835,8 +1835,89 @@ HWTEST_F(CommandListTest, givenComputeCommandListWhenMemoryCopyWithReservedDevic EXPECT_EQ(ZE_RESULT_SUCCESS, res); } +HWTEST_F(CommandListTest, givenPlainDeviceAllocationWhenQueryingPeerReservationBlockEndThenWholeRangeIsReturned) { + auto commandList = std::make_unique>>(); + commandList->initialize(device, NEO::EngineGroupType::renderCompute, 0u); + + void *ptr = nullptr; + size_t size = MemoryConstants::pageSize64k; + ze_device_mem_alloc_desc_t deviceDesc = {}; + ASSERT_EQ(ZE_RESULT_SUCCESS, context->allocDeviceMem(device->toHandle(), &deviceDesc, size, 0u, &ptr)); + + // not a virtual reservation, so the copy must not be split + EXPECT_EQ(size, commandList->bytesToPeerReservationBlockEnd(ptr, size)); + + EXPECT_EQ(ZE_RESULT_SUCCESS, context->freeMem(ptr)); +} + +HWTEST_F(CommandListTest, givenReservedDeviceAllocationOnSameDeviceWhenQueryingPeerReservationBlockEndThenWholeRangeIsReturned) { + auto commandList = std::make_unique>>(); + commandList->initialize(device, NEO::EngineGroupType::renderCompute, 0u); + + driverHandle->devices[0]->getNEODevice()->getExecutionEnvironment()->rootDeviceEnvironments[0]->memoryOperationsInterface = + std::make_unique(); + + void *buffer = nullptr; + size_t size = MemoryConstants::pageSize64k; + size_t reservationSize = size * 2; + + ASSERT_EQ(ZE_RESULT_SUCCESS, context->reserveVirtualMem(nullptr, reservationSize, &buffer)); + ze_physical_mem_desc_t desc = {}; + desc.size = size; + ze_physical_mem_handle_t phys0, phys1; + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(device->toHandle(), &desc, &phys0)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(device->toHandle(), &desc, &phys1)); + void *secondBlock = reinterpret_cast(reinterpret_cast(buffer) + size); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(buffer, size, phys0, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(secondBlock, size, phys1, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + + // the reservation is on the same device, so no peer import is involved and a copy + // spanning both blocks must not be split + EXPECT_EQ(reservationSize, commandList->bytesToPeerReservationBlockEnd(buffer, reservationSize)); + + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(buffer, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(secondBlock, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->freeVirtualMem(buffer, reservationSize)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys0)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys1)); +} + using MultiDeviceReservationCommandListTest = Test; +HWTEST_F(MultiDeviceReservationCommandListTest, givenReservedAllocationOwnedByAnotherDeviceWhenQueryingPeerReservationBlockEndThenRangeIsClampedToBlockEnd) { + auto deviceOwner = driverHandle->devices[0]; + auto devicePeer = driverHandle->devices[1]; + + auto commandList = std::make_unique>>(); + commandList->initialize(devicePeer, NEO::EngineGroupType::renderCompute, 0u); + + void *buffer = nullptr; + size_t size = MemoryConstants::pageSize64k; + size_t reservationSize = size * 2; + + ASSERT_EQ(ZE_RESULT_SUCCESS, context->reserveVirtualMem(nullptr, reservationSize, &buffer)); + ze_physical_mem_desc_t desc = {}; + desc.size = size; + ze_physical_mem_handle_t phys0, phys1; + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(deviceOwner->toHandle(), &desc, &phys0)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->createPhysicalMem(deviceOwner->toHandle(), &desc, &phys1)); + void *secondBlock = reinterpret_cast(reinterpret_cast(buffer) + size); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(buffer, size, phys0, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + ASSERT_EQ(ZE_RESULT_SUCCESS, context->mapVirtualMem(secondBlock, size, phys1, 0, ZE_MEMORY_ACCESS_ATTRIBUTE_READWRITE)); + + // each block is imported on the peer at an address of its own, so a range that + // crosses the block end is clamped there and copied in more than one step + EXPECT_EQ(size, commandList->bytesToPeerReservationBlockEnd(buffer, reservationSize)); + // a range that already fits inside one block is left alone + EXPECT_EQ(size, commandList->bytesToPeerReservationBlockEnd(buffer, size)); + + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(buffer, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->unMapVirtualMem(secondBlock, size)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->freeVirtualMem(buffer, reservationSize)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys0)); + EXPECT_EQ(ZE_RESULT_SUCCESS, context->destroyPhysicalMem(phys1)); +} + HWTEST_F(MultiDeviceReservationCommandListTest, givenPeerAccessSpanningReservationBlocksWhenAppendingMemoryCopyThenCopyIsRejectedInsteadOfHanging) { auto deviceOwner = driverHandle->devices[0]; auto devicePeer = driverHandle->devices[1];