Migrate amdgpu-windows-interop to rocm-systems (#808)

This commit is contained in:
Joseph Macaranas
2025-09-05 10:32:44 -04:00
committed by GitHub
parent 3d9d35a1f8
commit 5ca7af2d30
261 changed files with 86831 additions and 2 deletions
@@ -0,0 +1,212 @@
/*
***********************************************************************************************************************
*
* Copyright (c) 2023-2025 Advanced Micro Devices, Inc. All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a copy
* of this software and associated documentation files (the "Software"), to deal
* in the Software without restriction, including without limitation the rights
* to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
* copies of the Software, and to permit persons to whom the Software is
* furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
* SOFTWARE.
*
**********************************************************************************************************************/
#pragma once
#include "palGpaSession.h"
#include "palGpuUtil.h"
#include "palTraceSession.h"
#include "palVector.h"
#include "palHashSet.h"
#include "palMutex.h"
namespace Pal
{
class IPlatform;
class IDevice;
class IShaderLibrary;
} // namespace Pal
namespace GpuUtil
{
class GpaSession;
} // namespace GpuUtil
namespace GpuUtil
{
namespace TraceChunk
{
/// "CodeObject" RDF chunk identifier & version
constexpr char CodeObjectChunkId[TextIdentifierSize] = "CodeObject";
constexpr Pal::uint32 CodeObjectChunkVersion = 2;
/// Header for the "CodeObject" RDF chunk
struct CodeObjectHeader
{
Pal::uint32 pciId; /// The ID of the GPU the trace was run on
Pal::ShaderHash codeObjectHash; /// Hash of the Code Object binary
};
/// "COLoadEvent" RDF chunk identifier & version
constexpr char CodeObjectLoadEventChunkId[TextIdentifierSize] = "COLoadEvent";
constexpr Pal::uint32 CodeObjectLoadEventChunkVersion = 3;
struct CodeObjectLoadEventHeader
{
Pal::uint32 count; /// Number of load events in this chunk
};
/// Describes whether a load event was into GPU memory or from.
enum class CodeObjectLoadEventType : Pal::uint32
{
LoadToGpuMemory = 0, /// Code Object was loaded into GPU memory
UnloadFromGpuMemory = 1 /// Code Object was unloaded from GPU memory
};
/// Describes one or more GPU load/unload(s) of a Code Object. Payload for "COLoadEvent" RDF chunk.
struct CodeObjectLoadEvent
{
Pal::uint32 pciId; /// The ID of the GPU the trace was run on
CodeObjectLoadEventType eventType; /// Type of loader event
Pal::uint64 baseAddress; /// Base address where the Code Object was loaded
Pal::ShaderHash codeObjectHash; /// Hash of the (un)loaded Code Object binary
Pal::uint64 timestamp; /// CPU timestamp of this event being triggered
};
/// "PsoCorrelation" RDF chunk identifier & version
constexpr char PsoCorrelationChunkId[TextIdentifierSize] = "PsoCorrelation";
constexpr Pal::uint32 PsoCorrelationChunkVersion = 3;
struct PsoCorrelationHeader
{
Pal::uint32 count; /// Number of PSO correlations in this chunk
};
/// Payload for the "PsoCorrelation" RDF chunks
struct PsoCorrelation
{
Pal::uint32 pciId; /// The ID of the GPU the trace was run on
Pal::uint64 apiPsoHash; /// Hash of the API-level Pipeline State Object
Pal::PipelineHash internalPipelineHash; /// Hash of all inputs to the pipeline compiler
char apiLevelObjectName[64]; /// Debug object name (null-terminated)
};
/// "COCorrelation" RDF chunk identifier & version
constexpr char CodeObjectCorrelationChunkId[TextIdentifierSize] = "COCorrelation";
constexpr uint32_t CodeObjectCorrelationChunkVersion = 4;
struct CodeObjectCorrelationHeader
{
Pal::uint32 count; /// Number of Code Object Correlations in this chunk
};
/// Payload for the "CodeObjectCorrelation" RDF chunks
struct CodeObjectCorrelation
{
Pal::PipelineHash internalPipelineHash; /// Hash of all inputs to the pipeline compiler
Pal::ShaderHash codeObjectHash; /// Hash of the Code Object binary in the CO Database
Pal::uint32 containsMetadata : 1; /// 1 if the code object contains metadata, 0 otherwise
Pal::uint32 reserved : 31; /// Bitflags reserved for future use
};
} // namespace TraceChunk
/// CodeObject Trace Source name & version
constexpr char CodeObjectTraceSourceName[] = "codeobject";
constexpr Pal::uint32 CodeObjectTraceSourceVersion = 3;
// =====================================================================================================================
class CodeObjectTraceSource : public ITraceSource
{
public:
CodeObjectTraceSource(Pal::IPlatform* pPlatform);
~CodeObjectTraceSource();
// ==== TraceSource Native Functions ========================================================================== //
Pal::Result RegisterPipeline(const Pal::IPipeline* pPipeline, const RegisterPipelineInfo& clientInfo);
Pal::Result UnregisterPipeline(const Pal::IPipeline* pPipeline);
Pal::Result RegisterLibrary(const Pal::IShaderLibrary* pLibrary, const RegisterLibraryInfo& clientInfo);
Pal::Result UnregisterLibrary(const Pal::IShaderLibrary* pLibrary);
Pal::Result RegisterElfBinary(const ElfBinaryInfo& elfBinaryInfo);
Pal::Result UnregisterElfBinary(const ElfBinaryInfo& elfBinaryInfo);
// ==== Base Class Overrides =================================================================================== //
virtual void OnConfigUpdated(DevDriver::StructuredValue* pJsonConfig) override { }
virtual Pal::uint64 QueryGpuWorkMask() const override { return 0; }
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 908
virtual void OnTraceAccepted(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) override { }
#else
virtual void OnTraceAccepted() override { }
#endif
virtual void OnTraceBegin(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) override { }
virtual void OnTraceEnd(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) override { }
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 939
virtual void OnPostambleEnd(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer* pCmdBuf) override { }
#endif
virtual void OnTraceFinished() override;
virtual const char* GetName() const override { return CodeObjectTraceSourceName; }
virtual Pal::uint32 GetVersion() const override { return CodeObjectTraceSourceVersion; }
private:
Pal::Result RegisterSinglePipeline(const Pal::IPipeline* pPipeline, const RegisterPipelineInfo& clientInfo);
Pal::Result UnregisterSinglePipeline(const Pal::IPipeline* pPipeline);
Pal::Result AddCodeObjectLoadEvent(
const Pal::IShaderLibrary* pLibrary,
TraceChunk::CodeObjectLoadEventType eventType);
Pal::Result AddCodeObjectLoadEvent(
const Pal::IPipeline* pLibrary,
TraceChunk::CodeObjectLoadEventType eventType);
Pal::Result AddCodeObjectLoadEvent(
const ElfBinaryInfo& elfBinaryInfo,
TraceChunk::CodeObjectLoadEventType eventType);
Pal::Result WriteCodeObjectChunks();
Pal::Result WriteLoaderEventsChunk();
Pal::Result WritePsoCorrelationChunk();
Pal::Result WriteCoCorrelationChunk();
struct CodeObjectDatabaseRecord
{
Pal::uint32 recordSize;
Pal::ShaderHash codeObjectHash;
};
Pal::IPlatform* const m_pPlatform;
Util::RWLock m_registerPipelineLock;
Util::Vector<CodeObjectDatabaseRecord*, 1, Pal::IPlatform> m_codeObjectRecords;
Util::Vector<TraceChunk::CodeObjectLoadEvent, 1, Pal::IPlatform> m_loadEventRecords;
Util::Vector<TraceChunk::PsoCorrelation, 1, Pal::IPlatform> m_psoCorrelationRecords;
Util::Vector<TraceChunk::CodeObjectCorrelation, 1, Pal::IPlatform> m_coCorrelationRecords;
// API hashes -> internal pipeline hash (-> child code object hashes)
Util::HashSet<Pal::uint64, Pal::IPlatform, Util::JenkinsHashFunc> m_registeredApiHashes;
Util::HashSet<Pal::uint64, Pal::IPlatform, Util::JenkinsHashFunc> m_registeredPipelines;
Util::HashSet<Pal::uint64, Pal::IPlatform, Util::JenkinsHashFunc> m_registeredCoHashes;
};
} // namespace GpuUtil
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,141 @@
/*
***********************************************************************************************************************
*
* Copyright (c) 2014-2025 Advanced Micro Devices, Inc. All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a copy
* of this software and associated documentation files (the "Software"), to deal
* in the Software without restriction, including without limitation the rights
* to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
* copies of the Software, and to permit persons to whom the Software is
* furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
* SOFTWARE.
*
**********************************************************************************************************************/
/**
***********************************************************************************************************************
* @file palGpuUtil.h
* @brief Common include for the PAL GPU utility collection. Defines common types, macros, enums, etc.
***********************************************************************************************************************
*/
#pragma once
#include "pal.h"
// Forward declarations.
namespace Pal
{
struct DeviceProperties;
class IImage;
class IGpuMemory;
struct ImageCopyRegion;
struct TypedBufferCopyRegion;
struct MemoryImageCopyRegion;
}
/// Library-wide namespace encapsulating all PAL GPU utility entities.
namespace GpuUtil
{
/// Validate image copy region.
///
/// @param [in] properties The device properties.
/// @param [in] engineType Engine to validate.
/// @param [in] src Src image.
/// @param [in] dst Des image.
/// @param [in] region Copy region.
///
/// @returns true if the image copy is supported by the specific engine, otherwise false.
extern bool ValidateImageCopyRegion(
const Pal::DeviceProperties& properties,
Pal::EngineType engineType,
const Pal::IImage& src,
const Pal::IImage& dst,
const Pal::ImageCopyRegion& region);
/// Validate typed buffer copy region.
///
/// @param [in] properties The device properties.
/// @param [in] engineType Engine to validate.
/// @param [in] region Copy region.
///
/// @returns true if the typed buffer copy is supported by the specific engine, otherwise false.
extern bool ValidateTypedBufferCopyRegion(
const Pal::DeviceProperties& properties,
Pal::EngineType engineType,
const Pal::TypedBufferCopyRegion& region);
/// Validate image-memory copy region.
///
/// @param [in] properties The device properties.
/// @param [in] engineType Engine to validate.
/// @param [in] image The IImage object.
/// @param [in] region Copy region.
///
/// @returns true if the image-memory copy is supported by the specific engine, otherwise false.
extern bool ValidateMemoryImageRegion(
const Pal::DeviceProperties& properties,
Pal::EngineType engineType,
const Pal::IImage& image,
const Pal::IGpuMemory& memory,
const Pal::MemoryImageCopyRegion& region);
/// Generate a 64-bit uniqueId for a GPU memory allocation
///
/// @param [in] isInterprocess Indicates this uniqueId is for an externally shareable GPU memory allocation
///
/// @returns 64-bit uniqueId
extern Pal::uint64 GenerateGpuMemoryUniqueId(
bool isInterprocess);
} // GpuUtil
/**
***********************************************************************************************************************
* @page GpuUtilOverview GPU Utility Collection
*
* In addition to the generic, OS-abstracted software utilities, PAL provides GPU-specific utilities in the @ref GpuUtil
* namespace. The PAL GPU Utility Collection relies on both PAL core and PAL Utility. They are also available for use by
* its clients.
*
* All available PAL GPU utilities are defined in the @ref GpuUtil namespace, and are briefly summarized below. See the
* Reference topics for more detailed information on specific classes, enums, etc.
*
* ### TextWriter
* The TextWriter GPU utility class provides a method for clients to write text directly to an image. This can be used
* for debugging purposes. PAL's internal DbgOverlay uses the TextWriter class to write information about the current
* FPS and total allocated GPU video memory usage.
*
* The TextWriter class is broken up into palTextWriter.h and palTextWriterImpl.h. The intention is that palTextWriter.h
* will be included from other header files that need a full TextWriter definition, while palTextWriterImpl.h will be
* included by .cpp files that actually interact with the TextWriter. This should keep build times down versus putting
* all implementations directly in palTextWriter.h.
*
* Also included in the TextWriter is the TextWriterFont namespace, which defines the shader IL for drawing the text via
* a compute shader. It also defines the Font data, which is a packed binary that represents which pixels of a 10x16
* rectangle to render. The font is monospaced.
*
* ### Helper Functions
* ValidateImageCopyRegion - Validate the image copy region, returns true if the image copy is supported by the specific
* engine, otherwise false.
*
* ValidateTypedBufferCopyRegion - Validate the typed buffer copy region, returns true if the typed buffer copy is
* supported by the specific engine, otherwise false.
*
* ValidateMemoryImageRegion - Validate the image-memory copy region, returns true if the image-memory copy is supported
* by the specific engine, otherwise false.
*
* Next: @ref Overview
***********************************************************************************************************************
*/
@@ -0,0 +1,236 @@
/*
***********************************************************************************************************************
*
* Copyright (c) 2024-2025 Advanced Micro Devices, Inc. All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a copy
* of this software and associated documentation files (the "Software"), to deal
* in the Software without restriction, including without limitation the rights
* to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
* copies of the Software, and to permit persons to whom the Software is
* furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
* SOFTWARE.
*
**********************************************************************************************************************/
#pragma once
#include "palGpuUtil.h"
#include "palTraceSession.h"
#include "palGpaSession.h"
#include <atomic>
struct SqttQueueEventRecord;
struct SqttQueueInfoRecord;
namespace Pal
{
class Platform;
}
namespace GpuUtil
{
namespace TraceChunk
{
/// "QueueInfo" RDF chunk identifier & version
constexpr char QueueInfoChunkId[TextIdentifierSize] = "QueueInfo";
constexpr Pal::uint32 QueueInfoChunkVersion = 1;
/// Enum describing logical queue types
enum class QueueType : Pal::uint8
{
Unknown = 0,
Universal = 1,
Compute = 2,
Dma = 3,
Encode = 4,
Decode = 5,
Security = 6,
VideoProcessor = 7
};
/// Enum describing hardware engine types
enum class HwEngineType : Pal::uint8
{
Unknown = 0,
Universal = 1,
Compute = 2,
ExclusiveCompute = 3,
Dma = 4,
Decode = 5,
Encode = 6,
HighPriorityUniversal = 7,
HighPriorityGraphics = 8,
Security = 9,
Vpe = 10
};
/// Structure describing a queue's properties
struct QueueInfo
{
Pal::uint32 pciId; ///< The ID of the GPU queried
Pal::uint64 queueId; ///< API-specific queue ID
Pal::uint64 queueContext; ///< OS-level queue context value from Windows KMD to correlate with ETW data.
/// Only applicable to D3D on Windows; 0 otherwise.
QueueType queueType; ///< The logical queue type
HwEngineType engineType; ///< The hardware engine that the queue is mapped to
};
// ------------------------------------------------------------------------------------------- //
/// "QueueEvent" RDF chunk identifier & version
constexpr char QueueEventChunkId[TextIdentifierSize] = "QueueEvent";
constexpr Pal::uint32 QueueEventChunkVersion = 1;
/// The type of queue-level timings event
enum class QueueEventType : Pal::uint32
{
CmdBufSubmit = 0,
SignalSemaphore = 1,
WaitSemaphore = 2,
Present = 3
};
/// Structure describing a queue-level timings event
struct QueueEvent
{
Pal::uint32 pciId; ///< The ID of the GPU queried
Pal::uint64 queueId; ///< The API-specific queue ID which triggered the event
QueueEventType eventType; ///< The type of the queue-timing event
Pal::uint32 sqttCmdBufId; ///< [`CmdBufSubmit` only; 0 otherwise]
/// SQTT command buffer ID matching CmdBufStart user data marker
Pal::uint64 frameIndex; ///< [`CmdBufSubmit` & `Present` only; 0 otherwise]
/// Global frame index incremented for each "Present" call
Pal::uint32 submitSubIndex; ///< [`CmdBufSubmit` only; 0 otherwise]
/// Sub-index of event within submission.
/// When there is only one CmdBuffer per submission, `submitSubIndex` is 0.
/// When there are multiple command buffers per submission, `submitSubIndex`
/// is incremented by one for each command buffer within the submission.
Pal::uint64 apiEventId; ///< [`CmdBufSubmit`] API-specific command buffer ID signaled
/// [`SignalSemaphore`] API-specific semaphore ID signaled
/// [`WaitSemaphore`] API-specific semaphore ID waited on
/// [`Present`] N/A (set to 0)
Pal::uint64 cpuTimestamp; ///< CPU start timestamp of when this event is triggered in clock cycle units
Pal::uint64 gpuTimestamp1; ///< [`CmdBufSubmit`] GPU timestamp when the HW execution of command buffer began
/// [`SignalSemaphore`] GPU timestamp when the HW signaled the queue semaphore
/// [`WaitSemaphore`] GPU timestamp when HW finished waiting on the semaphore
/// [`Present`] GPU timestamp when HW processed the Present call
///
/// All timestamps are expressed in clock cycle units.
Pal::uint64 gpuTimestamp2; ///< [`CmdBufSubmit` only; 0 otherwise]
/// GPU timestamp when the HW execution of command buffer finished
};
} // namespace TraceChunk
// QueueTimings Trace Source name & version
constexpr char QueueTimingsTraceSourceName[] = "queuetimings";
constexpr Pal::uint32 QueueTimingsTraceSourceVersion = 2;
// =====================================================================================================================
// This trace source captures queue timings data through GPA session & produces "QueueInfo" and "QueueEvent" RDF chunks
class QueueTimingsTraceSource : public ITraceSource
{
public:
explicit QueueTimingsTraceSource(Pal::IPlatform* pPlatform);
virtual ~QueueTimingsTraceSource();
// ==== TraceSource Native Functions ========================================================================== //
Pal::Result Init(Pal::IDevice* pDevice);
Pal::Result RegisterTimedQueue(Pal::IQueue* pQueue,
Pal::uint64 queueId,
Pal::uint64 queueContext);
Pal::Result UnregisterTimedQueue(Pal::IQueue* pQueue);
Pal::Result TimedSubmit(Pal::IQueue* pQueue,
const Pal::MultiSubmitInfo& submitInfo,
const TimedSubmitInfo& timedSubmitInfo);
Pal::Result TimedSignalQueueSemaphore(Pal::IQueue* pQueue,
Pal::IQueueSemaphore* pQueueSemaphore,
const TimedQueueSemaphoreInfo& timedSignalInfo,
Pal::uint64 value = 0);
Pal::Result TimedWaitQueueSemaphore(Pal::IQueue* pQueue,
Pal::IQueueSemaphore* pQueueSemaphore,
const TimedQueueSemaphoreInfo& timedWaitInfo,
Pal::uint64 value = 0);
Pal::Result TimedQueuePresent(Pal::IQueue* pQueue,
const TimedQueuePresentInfo& timedPresentInfo);
Pal::Result ExternalTimedWaitQueueSemaphore(Pal::uint64 queueContext,
Pal::uint64 cpuSubmissionTimestamp,
Pal::uint64 cpuCompletionTimestamp,
const TimedQueueSemaphoreInfo& timedWaitInfo);
Pal::Result ExternalTimedSignalQueueSemaphore(Pal::uint64 queueContext,
Pal::uint64 cpuSubmissionTimestamp,
Pal::uint64 cpuCompletionTimestamp,
const TimedQueueSemaphoreInfo& timedSignalInfo);
bool IsTimingInProgress() const;
// ==== Base Class Overrides =================================================================================== //
virtual void OnConfigUpdated(DevDriver::StructuredValue* pJsonConfig) override { };
virtual Pal::uint64 QueryGpuWorkMask() const override { return 0; }
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 908
virtual void OnTraceAccepted(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) override;
#else
virtual void OnTraceAccepted() override;
#endif
virtual void OnTraceBegin(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) override { };
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 939
virtual void OnPostambleEnd(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer* pCmdBuf) override;
virtual void OnTraceEnd(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer* pCmdBuf) override {};
#else
virtual void OnTraceEnd(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer* pCmdBuf) override;
#endif
virtual void OnTraceFinished() override;
virtual const char* GetName() const override { return QueueTimingsTraceSourceName; }
virtual Pal::uint32 GetVersion() const override { return QueueTimingsTraceSourceVersion; }
private:
void WriteQueueInfoChunks(
const SqttQueueInfoRecord* pQueueInfoRecords,
size_t numQueueInfoRecords);
void WriteQueueEventChunks(
const SqttQueueInfoRecord* pQueueInfoRecords,
size_t numQueueInfoRecords,
const SqttQueueEventRecord* pQueueEventRecords,
size_t numQueueEventRecords);
void ReportInternalError(const char* pErrorMsg, Pal::Result result);
Pal::IPlatform* const m_pPlatform; // IPlatform owning the parent TraceSession
GpaSession* m_pGpaSession; // Handle to GpaSession object for tracking queue timings
bool m_traceIsHealthy; // Internal flag for tracking resource and state health
std::atomic<bool> m_timingInProgress; // Flag for tracking if queue timings operations are ongoing
};
} // namespace GpuUtil
@@ -0,0 +1,150 @@
/*
***********************************************************************************************************************
*
* Copyright (c) 2024-2025 Advanced Micro Devices, Inc. All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a copy
* of this software and associated documentation files (the "Software"), to deal
* in the Software without restriction, including without limitation the rights
* to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
* copies of the Software, and to permit persons to whom the Software is
* furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
* SOFTWARE.
*
**********************************************************************************************************************/
#pragma once
#include "palTraceSession.h"
namespace Pal
{
class IPlatform;
class IQueue;
class ICmdBuffer;
class Device;
}
namespace GpuUtil
{
/// Supported render operations used to advance the trace
enum RenderOp : Pal::uint8
{
RenderOpDraw = (1u << 0),
RenderOpDispatch = (1u << 1)
};
/// Structure used to batch submit render operations on queue submission
/// This struct should have a `*Count` field for each @ref RenderOp enumeration above
struct RenderOpCounts
{
Pal::uint32 drawCount;
Pal::uint32 dispatchCount;
};
constexpr Pal::uint32 RenderOpTraceControllerVersion = 4;
constexpr char RenderOpTraceControllerName[] = "renderop";
// =====================================================================================================================
class RenderOpTraceController : public ITraceController
{
public:
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION < 896
using RenderOp = GpuUtil::RenderOp;
#endif
RenderOpTraceController(Pal::IPlatform* pPlatform, Pal::IDevice* pDevice);
virtual ~RenderOpTraceController();
virtual const char* GetName() const override { return RenderOpTraceControllerName; }
virtual Pal::uint32 GetVersion() const override { return RenderOpTraceControllerVersion; }
virtual void OnConfigUpdated(DevDriver::StructuredValue* pJsonConfig) override;
virtual Pal::Result OnTraceRequested() override;
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 908
virtual Pal::Result OnPreparationGpuWork(Pal::uint32 gpuIndex, Pal::ICmdBuffer** ppCmdBuf) override;
#endif
virtual Pal::Result OnBeginGpuWork(Pal::uint32 gpuIndex, Pal::ICmdBuffer** ppCmdBuffer) override;
virtual Pal::Result OnEndGpuWork(Pal::uint32 gpuIndex, Pal::ICmdBuffer** ppCmdBuffer) override;
virtual Pal::Result OnEndPostambleGpuWork(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer** ppCmdBuffer) override;
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION < 896
void RecordRenderOp(Pal::IQueue* pQueue, RenderOp renderOp);
#endif
void FinishTrace();
// Cancel the trace currently in progress.
virtual Pal::Result OnTraceCanceled() override;
/// This function must be called by client drivers implementing the RenderOp controller.
/// On every queue submission, this function is called with the cumulative counts of render operations
/// recorded into that queue's command buffers.
/// Based on the controller's internal mask, set by the user during trace configuration,
/// the trace controller may advance its state.
void RecordRenderOps(Pal::IQueue* pQueue, const RenderOpCounts& renderOpCounts);
private:
/// Controls whether the trace proceeds on absolute render op counts or relative
enum class CaptureMode : Pal::uint8
{
Relative = 0, ///< Relative to when the trace request is received
Absolute ///< Absolute render op index
};
Pal::Result AcceptTrace();
Pal::Result BeginTrace();
Pal::Result SubmitBeginTraceGpuWork() const;
Pal::Result SubmitEndTraceGpuWork();
Pal::Result SubmitEndPostambleGpuWork();
Pal::Result WaitForTraceEndGpuWorkCompletion() const;
Pal::Result CreateFence(Pal::IFence** ppFence) const;
Pal::Result CreateCommandBuffer(bool traceEnd, Pal::ICmdBuffer** ppCmdBuf) const;
Pal::Result CreateCmdAllocator();
void OnRenderOpUpdated(Pal::uint64 countRecorded);
void FreeResources();
void AbortTrace();
Pal::IPlatform* const m_pPlatform; // Platform associated with this TraceController
Pal::IDevice* m_pDevice; // Device associated with this TraceController
Pal::ICmdAllocator* m_pCmdAllocator; // Command allocator for the TraceController
TraceSession* m_pTraceSession; // TraceSession owning this TraceController
Pal::uint64 m_supportedGpuMask; // Bit mask of GPU indices that are capable of participating in the trace
Pal::uint8 m_renderOpMask; // Bitmask of RenderOp modes, indicating which are accepted
CaptureMode m_captureMode; // Modality for determining the starting renderop index of the trace
Pal::uint64 m_renderOpCount; // The "global" count, incremented on every render op
Pal::uint64 m_prepStartRenderOp; // Relative or absolute render op number indicating trace begin
Pal::uint64 m_numPrepRenderOps; // Number of "warm-up" frames before the start frame
Pal::uint64 m_captureRenderOpCount; // Number of frames to wait before ending the trace
Pal::uint64 m_renderOpTraceAccepted; // The frame number when the trace was accepted
Util::Mutex m_renderOpLock; // Lock over UpdateFrame/OnFrameUpdated
Pal::IQueue* m_pQueue; // The queue being used to submit Begin/End GPU trace command buffers
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 908
Pal::ICmdBuffer* m_pCmdBufTracePrepare; // Command buffer for recording during the prep phase
#endif
Pal::ICmdBuffer* m_pCmdBufTraceBegin; // Command buffer to submit Trace Begin
Pal::ICmdBuffer* m_pCmdBufTraceEnd; // Command buffer to submit Trace End
Pal::ICmdBuffer* m_pCmdBufPostambleEnd; // Command buffer to submit Postamble End
Pal::IFence* m_pFenceTraceEnd; // Fence to wait for Trace End command buffer completion
Pal::IFence* m_pFencePostambleEnd; // Fence to wait for Postamble End command buffer completion
};
} // namespace GpuUtil
@@ -0,0 +1,737 @@
/*
***********************************************************************************************************************
*
* Copyright (c) 2021-2025 Advanced Micro Devices, Inc. All Rights Reserved.
*
* Permission is hereby granted, free of charge, to any person obtaining a copy
* of this software and associated documentation files (the "Software"), to deal
* in the Software without restriction, including without limitation the rights
* to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
* copies of the Software, and to permit persons to whom the Software is
* furnished to do so, subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
* FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
* AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
* LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
* OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
* SOFTWARE.
*
**********************************************************************************************************************/
/**
***********************************************************************************************************************
* @file palTraceSession.h
* @brief PAL GPU utility TraceSession class.
***********************************************************************************************************************
*/
#pragma once
#include "palPlatform.h"
#include "palDeque.h"
#include "palDevice.h"
#include "palGpuUtil.h"
#include "palHashMap.h"
#include "palMutex.h"
#include "palPipeline.h"
#include "palSysMemory.h"
#include "palGpuMemory.h"
#include "palMemTrackerImpl.h"
#include "palVector.h"
struct rdfStream;
struct rdfChunkFileWriter;
namespace DevDriver
{
class IStructuredWriter;
class IStructuredReader;
class StructuredValue;
}
namespace GpuUtil
{
class ITraceController;
class ITraceSource;
constexpr Pal::uint16 TextIdentifierSize = 16;
/// Information required to create a new chunk of trace data in a TraceSession
///
/// This data inside this structure is expected to be produced by trace source implementations. The specific fields
/// included within this structure are intended to support compatibility with the Radeon Data Format (RDF) spec.
struct TraceChunkInfo
{
char id[TextIdentifierSize]; ///< Text identifier of the chunk
Pal::uint32 version; ///< Version number of the chunk
const void* pHeader; ///< [in] Pointer to a buffer that contains the header data for the chunk
Pal::int64 headerSize; ///< Size of the buffer pointed to by pHeader
const void* pData; ///< [in] Pointer to a buffer that contains the data for the chunk
Pal::int64 dataSize; ///< Size of the buffer pointed to by pData
bool enableCompression; ///< Indicates if the chunk's data should be compressed or not
};
/// The available states of TraceSession
enum class TraceSessionState : Pal::uint32
{
Ready = 0, ///< New trace ready to begin
Requested = 1, ///< A trace has been requested and awaiting acceptance
Preparing = 2, ///< Trace has been accepted and is preparing resources before beginning
Running = 3, ///< Trace is in progress
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 939
Postamble = 4, ///< The detailed frame trace has ended but its data has not yet been written
/// into the session. Some trace sources may still collect data during this time.
PostambleWaiting = 5, ///< Waiting for Postamble to complete.
Completed = 6, ///< Trace has fully completed. RDF trace data is ready to be pulled out by CollectTrace().
Count = 7
#else
Waiting = 4, ///< Trace has ended, but data has not been written into the session
Completed = 5, ///< Trace has fully completed. RDF trace data is ready to be pulled out by CollectTrace().
Count = 6
#endif
};
/// Defines the type of payload. Currently only strings are supported but in the future can include JSON, structs, etc.
enum class TraceErrorPayload : Pal::uint32
{
None, //< Should be set when there is no additional information to be sent with the error
ErrorString //< Should be set when the error payload is string data
};
/// Chunk header for the error tracing chunk
struct TraceErrorHeader
{
char chunkId[TextIdentifierSize]; ///< Text identifier of the failing chunk
Pal::uint32 chunkIndex; ///< Chunk index of the failing chunk
Pal::Result resultCode; ///< PAL Result code of the failure
TraceErrorPayload payloadType; ///< Type of error chunk payload
};
constexpr char ErrorChunkTextIdentifier[TextIdentifierSize] = "TraceError";
constexpr Pal::uint32 ErrorTraceChunkVersion = 1;
/**
***********************************************************************************************************************
* @interface ITraceController
* @brief Interface that allows for control of a trace operation through TraceSession.
*
* Trace controllers are responsible for driving the high-level steps of a trace operation. Users of this interface are
* expected to create their own implementation of this interface, register it with a TraceSession, then call the
* following TraceSession functions to drive the trace process:
*
* TraceSession::AcceptTrace
* TraceSession::BeginTrace
* TraceSession::EndTrace
* TraceSession::EndPostamble
* TraceSession::FinishTrace
***********************************************************************************************************************
*/
class ITraceController
{
public:
/// Returns the name of the controller
///
/// @returns the name of the controller as a null terminated string
virtual const char* GetName() const = 0;
/// Returns the version of the controller
///
/// @returns the version of the controller as an unsigned integer value
virtual Pal::uint32 GetVersion() const = 0;
/// Called by the associated session to update the current trace configuration
///
/// @param [in] pJsonConfig Configuration data formatted as json and stored as DevDriver's StructuredValue object
virtual void OnConfigUpdated(DevDriver::StructuredValue* pJsonConfig) = 0;
/// Called by the associated session to notify the controller that a trace has been requested and it can take
/// control of the TraceSession when desired.
virtual Pal::Result OnTraceRequested() = 0;
/// Called by the associated session to notify the controller that a trace has been canceled and it can start
/// canceling the trace when ready.
virtual Pal::Result OnTraceCanceled() = 0;
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 908
/// Called by TraceSession to indicate that GPU work is required on the indicated GPU during the preparation phase.
/// The command buffer must be ready to record commands; however, the trace controller should not submit it
/// until the trace begins.
///
/// The controller MUST return a valid command buffer that is ready to record commands for the target GPU
/// upon successful completion of this function via ppCmdBuf.
///
/// This function will be called once per trace for each GPU that's considered relevant by the current set of
/// trace sources.
///
/// Note: This command buffer should be submitted at the same time as the command buffer provided in
/// `OnBeginGpuWork`. They may be the same command buffer or separate; the goal is to allow trace sources
/// to frontload recording GPU work before the trace formally begins.
///
/// Note: The command buffer provided by this function does not need to be a new command buffer. It just needs
/// to be capable of recording new commands.
///
/// @param [in] gpuIndex The index of the target GPU
/// @param [out] ppCmdBuf A command buffer that can be used to record GPU work before a trace starts executing.
/// Note that this command buffer shouldn't be submitted until the trace begins.
///
/// @returns Success if the command buffer was successfully returned
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
virtual Pal::Result OnPreparationGpuWork(Pal::uint32 gpuIndex, Pal::ICmdBuffer** ppCmdBuf) = 0;
#endif
/// Called by TraceSession to indicate that GPU work is required to begin a trace on the indicated GPU
///
/// The controller MUST return a valid command buffer that is ready to record commands for the target GPU
/// upon successful completion of this function via ppCmdBuf.
///
/// This function will be called once per trace for each GPU that's considered relevant by the current set of
/// trace sources.
///
/// Note: The command buffer provided by this function does not need to be a new command buffer. It just needs
/// to be capable of recording new commands.
///
/// @param [in] gpuIndex The index of the target GPU
/// @param [out] ppCmdBuf A command buffer that can be used to perform any GPU work required to begin the trace
///
/// @returns Success if the command buffer was successfully returned
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
virtual Pal::Result OnBeginGpuWork(Pal::uint32 gpuIndex, Pal::ICmdBuffer** ppCmdBuf) = 0;
/// Called by TraceSession to indicate that GPU work is required to end a trace on the indicated GPU
///
/// The controller MUST return a valid command buffer that is ready to record commands for the target GPU
/// upon successful completion of this function via ppCmdBuf.
///
/// This function will be called once per trace for each GPU that's considered relevant by the current set of
/// trace sources.
///
/// Note: The command buffer provided by this function does not need to be a new command buffer. It just needs
/// to be capable of recording new commands.
///
/// @param [in] gpuIndex The index of the target GPU
/// @param [out] ppCmdBuf A command buffer that can be used to perform any GPU work required to end the trace
///
/// @returns Success if the command buffer was successfully returned
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
virtual Pal::Result OnEndGpuWork(Pal::uint32 gpuIndex, Pal::ICmdBuffer** ppCmdBuf) = 0;
/// Called by TraceSession to indicate that GPU work is required to end the postamble on the indicated GPU
///
/// The controller MUST return a valid command buffer that is ready to record commands for the target GPU
/// upon successful completion of this function via ppCmdBuf.
///
/// This function will be called once per trace for each GPU that's considered relevant by the current set of
/// trace sources.
///
/// Note: The command buffer provided by this function does not need to be a new command buffer. It just needs
/// to be capable of recording new commands.
///
/// @param [in] gpuIndex The index of the target GPU
/// @param [out] ppCmdBuf A command buffer that can be used to perform any GPU work required to end the postamble
///
/// @returns Success if the command buffer was successfully returned
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
virtual Pal::Result OnEndPostambleGpuWork(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer** ppCmdBuf) = 0;
};
/**
***********************************************************************************************************************
* @interface ITraceSource
* @brief Interface that enables developers to emit arbitrary data chunks into a trace through TraceSession.
*
* Trace sources are used to implement any surrounding logic required to produce a trace data chunk. Users of this
* interface are expected to create their own implementation of this interface, register it with a TraceSession, then
* call TraceSession::WriteDataChunk during a trace operation whenever a data chunk should be produced.
***********************************************************************************************************************
*/
class ITraceSource
{
public:
/// Called by the associated session to update the current trace configuration
///
/// @param [in] pJsonConfig Configuration data formatted as json and stored as DevDriver's StructuredValue object
virtual void OnConfigUpdated(DevDriver::StructuredValue* pJsonConfig) = 0;
/// Returns a bitmask that represents which GPUs are relevant to this trace source
///
/// If the bit at index N is set, GPU N must execute work on the GPU in order to produce trace data
virtual Pal::uint64 QueryGpuWorkMask() const = 0;
/// Called by the associated session to notify the source that a new trace has been accepted
///
/// The source may use this notification to do any preparation work that might be required before the trace begins.
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 908
/// A command buffer is provided for the trace source to insert any work into. Note that the work will not be
/// submitted until the trace begins (at the same time as `OnTraceBegin`). This allows for frontloading of
/// expensive operations, such as the construction of a GpaSession sample, that would affect runtime speed
/// or behavior during trace exeecution.
///
/// @param [in] gpuIndex The index of the GPU that owns pCmdBuf
/// @param [in] pCmdBuf A command buffer that can be used to record any GPU work required during the
/// preparation phase of the trace. Not submitted until `OnTraceBegin`.
virtual void OnTraceAccepted(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) = 0;
#else
virtual void OnTraceAccepted() = 0;
#endif
/// Called by the associated session to notify the source that it should begin a trace
///
/// The source should use the provided command buffer to execute any GPU work that's required for the source to
/// begin a trace operation.
///
/// In situations where multiple GPUs are present, this function will be called for all GPUs that are expected to
/// participate in the trace. All GPUs that begin a trace are required to end it later. Sources are not expected
/// to handle cases where the begin/end function calls are mismatched during a trace operation.
///
/// @param [in] gpuIndex The index of the GPU that owns pCmdBuf
/// @param [in] pCmdBuf A command buffer that can be used to perform any GPU work required to begin the trace
virtual void OnTraceBegin(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) = 0;
/// Called by the associated session to notify the source that it should end the current trace
///
/// The source should use the provided command buffer to execute any GPU work that's required for the source to
/// end a trace operation.
///
/// The command buffer associated with the OnTraceBegin function is not guaranteed to have finished GPU execution
/// when this function is called. The command buffer associated with this function is also not guaranteed to finish
/// execution until OnTraceFinished is called.
///
/// In situations where multiple GPUs are present, this function will be called for all GPUs that are expected to
/// participate in the trace.
///
/// @param [in] gpuIndex The index of the GPU that owns pCmdBuf
/// @param [in] pCmdBuf A command buffer that can be used to perform any GPU work required to end the trace
virtual void OnTraceEnd(Pal::uint32 gpuIndex, Pal::ICmdBuffer* pCmdBuf) = 0;
/// Called by the associated session to notify the source that it should end the postamble
///
/// The source should use the provided command buffer to execute any GPU work that's required for the source to
/// end its postamble operation.
///
/// The command buffer associated with the OnTraceBegin and OnTraceEnd functions are not guaranteed to have
/// finished GPU execution when this function is called. The command buffer associated with this function is also
/// not guaranteed to finish execution until OnTraceFinished is called.
///
/// In situations where multiple GPUs are present, this function will be called for all GPUs that are expected to
/// participate in the trace.
///
/// @param [in] gpuIndex The index of the GPU that owns pCmdBuf
/// @param [in] pCmdBuf A command buffer that can be used to perform any GPU work required to end the postamble
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 939
virtual void OnPostambleEnd(
Pal::uint32 gpuIndex,
Pal::ICmdBuffer* pCmdBuf) = 0;
#endif
/// Called by the associated session to notify the source that the current trace has finished
///
/// When this function is called, all prior command buffers provided to the source during the trace operation have
/// finished execution. The source should use this function to collect any data generated by the GPU and emit it
/// via TraceSession::WriteDataChunk.
virtual void OnTraceFinished() = 0;
/// Returns the name of the source
///
/// @returns the name of the source as a null terminated string
virtual const char* GetName() const = 0;
/// Returns the version of the source
///
/// @returns the version of the source as an unsigned integer value
virtual Pal::uint32 GetVersion() const = 0;
/// Whether multiple instances of the trace source are allowed
///
/// @returns true if multiple instances of this trace sources can co-exist in one session, false otherwise.
virtual bool AllowMultipleInstances() const { return false; }
};
/**
***********************************************************************************************************************
* @class TraceSession
* @brief Helper class providing common driver functionality for collecting arbitrary data traces.
*
* Due to the global nature of the trace functionality, only one TraceSession is typically used at a time.
* An interface to acquire a session exists on IPlatform. Users who need to interact with an instance of this object
* should expect to acquire it there.
*
* @see IPlatform::GetTraceSession()
***********************************************************************************************************************
*/
class TraceSession final
{
public:
/// Constructor.
///
/// @param [in] pPlatform Platform associated with this TraceSesion
TraceSession(Pal::IPlatform* pPlatform);
/// Destructor
~TraceSession();
/// Initialize the trace session before requesting a trace.
///
/// @returns Success if initalization was successful, or ErrorUnknown upon failure.
Pal::Result Init();
/// Returns whether tracing has been formally enabled via UberTrace or not.
/// If 'true', this means that tool-side applications have requested this
/// TraceSession to capture traces. This has implications for PAL clients.
///
/// @returns True if tracing has been enabled, and false otherwise.
bool IsTracingEnabled() const { return m_tracingEnabled; }
/// Attempts to update the current trace configuration
///
/// This function will only succeed if there is currently to trace in progress
///
/// TODO: The JSON configuration interface will likely be replaced with driver settings in the future
///
/// @param [in] pData Buffer that stores the Json-formatted configuration data
/// @param [in] dataSize Configuration data-size
///
/// @returns Success if the trace configuration was successfully updated.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
/// + ErrorUnavailable if a trace is currently in progress
/// + ErrorInvalidPointer pData is nullptr
/// + ErrorInvalidParameter pData is not valid json
Pal::Result UpdateTraceConfig(const void* pData, size_t dataSize);
/// Attempts to request a new trace operation on the trace session.
///
/// Once a trace is successfully requested, it will become available for a registered trace controller to accept.
/// When a controller accepts the trace, it becomes responsible for managing the rest of the trace operation and
/// notifying the session upon trace completion.
///
/// Since the session can only run a single trace at a time, this function will not succeed if another trace is
/// is already requested or in progress.
///
/// @returns Success if the trace operation was successfully requested.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
/// + ErrorUnavailable if there is a trace in progress already and a new one cannot be started
Pal::Result RequestTrace();
/// Cancels a trace currently in progress.
///
/// @returns Success if the trace was successfully canceled.
/// Otherwise, one of the following errors may be returned:
/// + NotReady if the trace is not ready to be canceled.
/// + ErrorUnknown if an internal PAL error occurs.
Pal::Result CancelTrace();
/// Cleans up the RDF chunk stream and makes it ready for a new trace again.
///
/// @returns Success if the trace session and rdf streams were successfully cleaned up and returned to the
/// initialization state
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
Pal::Result CleanupChunkStream();
/// Attempts to consume any trace data stored within the trace session.
///
/// This function will only successfully return trace data after a trace operation is completed on the session.
///
/// TODO: This function should be replaced with one that uses a callback so we can avoid needing to store the trace
/// data into memory twice.
///
/// @param [out] pData (Optional) Destination buffer to copy the trace data into
/// If this parameter is nullptr, the size of the trace data in bytes will be
/// returned via pDataSize instead of consuming any trace data.
/// @param [in/out] pDataSize If pData is nullptr, then this parameter is used to return the trace data
/// size in bytes.
/// If pData is valid, this parameter represents the size of the buffer
/// pointed to by pData.
///
/// @returns Success if the trace data was successfully consumed or the size of the trace data was returned.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal error occurs in PAL or an unknown error is thrown by external library
/// + ErrorUnavailable if trace data is not available for collection at this time
/// + ErrorInvalidPointer if nullptr is passed as pDataSize
/// + ErrorInvalidMemorySize if *pDataSize indicates that pData is too small to contain the trace data
Pal::Result CollectTrace(void* pData, size_t* pDataSize);
/// Attempts to register a trace controller
///
/// Once registered, trace controllers can receive configuration updates from the session.
/// They may also manage the trace operation by calling AcceptTrace, BeginTrace, EndTrace, EndPostamble and FinishTrace.
///
/// Trace controllers can only be registered when there is no trace in progress
///
/// @param [in] pController The trace controller to register with the session
///
/// @returns Success if the controller was successfully registered.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
/// + AlreadyExists if this controller has already been registered
/// + ErrorUnavailable if a trace is in progress
/// + ErrorInvalidPointer if nullptr is passed as pController
Pal::Result RegisterController(ITraceController* pController);
/// Attempts to unregister a previously registered trace controller
///
/// @param [in] pController The trace controller to unregister from the session
///
/// @returns Success if the controller was successfully unregistered.
/// Otherwise, one of the following errors may be returned:
/// + NotFound if the provided controller was not previously registered
/// + ErrorUnknown if an internal PAL error occurs.
/// + ErrorUnavailable if a trace is in progress
Pal::Result UnregisterController(ITraceController* pController);
/// Attempts to register a trace source
///
/// Once registered, trace sources can receive configuration updates from the session.
/// They may also emit data during trace operations by calling WriteDataChunk.
///
/// Trace sources can only be registered when there is no trace in progress
///
/// @param [in] pSource The trace source to register with the session
///
/// @returns Success if the source was successfully registered.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
/// + AlreadyExists if this source has already been registered
/// + ErrorUnavailable if a trace is in progress
/// + ErrorInvalidPointer if nullptr is passed as pSource
Pal::Result RegisterSource(ITraceSource* pSource);
/// Attempts to unregister a previously registered trace source
///
/// @param [in] pSource The trace source to unregister from the session
///
/// @returns Success if the source was successfully unregistered.
/// Otherwise, one of the following errors may be returned:
/// + NotFound if the provided source was not previously registered
/// + ErrorUnknown if an internal PAL error occurs.
/// + ErrorUnavailable if a trace is in progress
Pal::Result UnregisterSource(ITraceSource* pSource);
/// Attempts to accept a previously requested trace with the provided controller
///
/// Once a trace is successfully accepted by a controller, that controller becomes responsible for managing the
/// rest of the trace operation. Also, once a requested trace is accepted by a controller, no other controllers
/// will be able to accept that trace. Accept is a "consuming" operation.
///
/// @param [in] pController The trace controller to accept the trace with
/// @param [in] supportedGpuMask Bit mask of GPU indices that are capable of participating in the trace
///
/// The GPU mask provided to this function is used to determine which GPUs will be involved in the trace. In order
/// to decide which GPUs require GPU work, the session creates a combined mask from all registered sources and
/// checks it against the mask provided by this function. Only GPUs that are present in both masks will be able to
/// submit GPU work during the trace.
///
/// @returns Success if the trace was successfully accepted.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal PAL error occurs.
/// + ErrorUnavailable if no trace has been requested or a trace is currently in progress
/// + ErrorInvalidPointer if nullptr is passed as pController
Pal::Result AcceptTrace(ITraceController* pController, Pal::uint64 supportedGpuMask);
/// Begins the trace that was previously accepted by the provided controller
///
/// This function MUST be called after a successful call to AcceptTrace. When this function is called, the session
/// will communicate with all registered trace sources and instruct them to begin the trace operation. The provided
/// trace controller will be notified if any GPU work is required via ITraceController::OnBeginGpuWork. The command
/// buffers returned by OnBeginGpuWork will be passed to each relevant trace source to record required work.
///
/// The command buffers generated in response to this this call MUST be submitted BEFORE the command buffers
/// generated in response to the EndTrace call!
///
/// In situations where multiple GPUs are present, the OnBeginGpuWork function will be called once per GPU index,
/// for all GPUs that are relevant for the current trace sources.
///
/// @returns Success if the trace was successfully started.
/// Otherwise, the error generated by OnBeginGpuWork will be returned.
Pal::Result BeginTrace();
/// Ends the trace that was previously started by the provided controller
///
/// This function MUST be called after BeginTrace. When this function is called, the session will communicate with
/// all registered trace sources and instruct them to end the trace operation. The provided trace controller will
/// trace controller will be notified if any GPU work is required via ITraceController::OnEndGpuWork. The command
/// buffers returned by OnEndGpuWork will be passed to each relevant trace source to record required work.
///
/// The command buffers generated in response to this this call MUST be submitted AFTER the command buffers
/// generated in response to the previous BeginTrace call! The generated command buffers MUST also complete
/// execution on the GPU BEFORE FinishTrace is called!
///
/// In situations where multiple GPUs are present, the OnEndGpuWork function will be called once per GPU index
/// for all GPUs that are relevant for the current trace sources.
///
/// The Trace Session will enter Postamble phase after EndTrace is called.
///
/// @returns Success if the trace was successfully ended.
/// Otherwise, the error generated by OnEndGpuWork will be returned.
Pal::Result EndTrace();
#if PAL_CLIENT_INTERFACE_MAJOR_VERSION >= 939
/// Ends the postamble phase, which typically runs until the detailed trace data is available.
/// This function MUST be called after EndTrace. When this function is called, the session will communicate with
/// all registered trace sources and notify them of the end of the postamble phase. The provided trace controller
/// will be notified if any GPU work is required via ITraceController::OnEndPostambleGpuWork. The command
/// buffers returned by OnEndPostambleGpuWork will be passed to each relevant trace source to record required work.
///
/// The command buffers generated in response to this this call MUST be submitted AFTER the command buffers
/// generated in response to the previous EndTrace call! The generated command buffers MUST also complete
/// execution on the GPU BEFORE FinishPostamble is called!
///
/// In situations where multiple GPUs are present, the OnEndPostambleGpuWork function will be called once per GPU index
/// for all GPUs that are relevant for the current trace sources.
///
/// @returns Success if the trace was successfully ended.
/// Otherwise, the error generated by OnEndPostambleGpuWork will be returned.
Pal::Result EndPostamble();
#endif
/// Notifies the session that the trace operation started by the provided controller has finished.
///
/// This function MUST be called after EndPostamble. When this function is called, the session will communicate with
/// all registered trace sources and notify them that all GPU work is complete. This notification is typically
/// used by sources to retrieve data produced by the GPU and write it into the session's trace data.
void FinishTrace();
/// Writes a chunk of trace data into the session.
///
/// Trace sources are expected to call this function whenever they produce a new data chunk that should be added
/// into the session's trace data.
///
/// This function may ONLY be called AFTER the BeginTrace function returns and BEFORE the FinishTrace call returns!
///
/// @param [in] pSource The trace source that generated the provided data chunk
/// @param [in] info Information about the provided chunk that will be written into the trace data
///
/// @returns Success if the incoming data chunk was successfully written/appended into the current data stream.
/// Otherwise, one of the following errors may be returned:
/// + ErrorUnknown if an internal error occurs in PAL or an unknown error is thrown by external library
Pal::Result WriteDataChunk(ITraceSource* pSource, const TraceChunkInfo& info);
/// Returns the current TraceSession state
///
/// @returns Enum value of the current TraceSessionState
TraceSessionState GetTraceSessionState() const
{
return m_sessionState;
}
/// Sets the TraceSession state based on external operations
///
/// @param [in] sessionState TraceSessionState value to be assigned as the current state
void SetTraceSessionState(TraceSessionState sessionState)
{
m_sessionState = sessionState;
}
/// Returns the current active controller
///
/// @returns Pointer to the current active controller driving the TraceSession
ITraceController* GetActiveController() const
{
return m_pActiveController;
}
/// Reports an error encountered during an active trace by inserting a "TraceError" chunk to the trace stream
///
/// If, during a trace or the construction of an RDF chunk, an error is encountered and a chunk that was
/// expected to be written can no longer be, this function may be called to insert an error chunk in place
/// of the expected chunk.
///
/// @param [in] chunkId Text identifier of the failed RDF chunk
/// @param [in] pPayload Pointer to the data sent for the error
/// If the payloadType is a string, the string must be null-terminated
/// @param [in] payloadSize Size of the data in the payload
/// @param [in] payloadType Type of payload data represented by `pPayload`
/// @param [in] errorResult The PAL result code of the encountered error
///
/// @returns Success if the error chunk was written successfully
Pal::Result ReportError(
const char chunkId[TextIdentifierSize],
const void* pPayload,
Pal::uint64 payloadSize,
TraceErrorPayload payloadType,
Pal::Result errorResult);
/// Explicitly activates this TraceSession for managing traces.
///
/// This should be called during Platform Init in response to a tool-side request to enable UberTrace tracing.
/// This signals that an active connection has been made to tool-side applications and that profiling via
/// PAL Trace should be prioritized in client drivers.
void EnableTracing()
{
m_tracingEnabled = true;
}
/// Returns a pointer to a byte array containing the trace configuration.
///
/// @param [out] pTraceConfigSize Sets *pTraceConfigSize to the number of bytes in the trace config
///
/// @returns A pointer to the trace configuration data
const void* GetTraceConfig(size_t* pTraceConfigSize) const
{
PAL_ASSERT(pTraceConfigSize != nullptr);
(*pTraceConfigSize) = m_configDataSize;
return m_pConfigData;
}
/// Indicates if a cancel-trace signal has been received and that a cancelation is in progress.
///
/// @return true if a cancelation is in progress.
bool IsCancelingTrace() const { return m_cancelingTrace; }
private:
typedef Pal::IPlatform TraceAllocator;
Pal::IPlatform* const m_pPlatform; // Platform associated with this TraceSesion
DevDriver::IStructuredReader* m_pReader; // Stores the current JSON-based config of the TraceSession
// RW Locks for trace sources, controllers, and RDF streams
Util::RWLock m_registerTraceSourceLock;
Util::RWLock m_registerTraceControllerLock;
Util::RWLock m_chunkAppendLock;
// Trace sources registered with this TraceSession.
using TraceSourcesVec = Util::Vector<ITraceSource*, 16, TraceAllocator>;
TraceSourcesVec m_registeredTraceSources;
// TraceSources and corresponding configs
typedef Util::HashMap <const char*,
DevDriver::StructuredValue*,
TraceAllocator,
Util::StringJenkinsHashFunc,
Util::StringEqualFunc> TraceSourcesConfigMap;
TraceSourcesConfigMap m_traceSourcesConfigs;
// Unique trace controllers registered with this TraceSession.
typedef Util::HashMap <const char*,
ITraceController*,
TraceAllocator,
Util::StringJenkinsHashFunc,
Util::StringEqualFunc> TraceControllersMap;
TraceControllersMap m_registeredTraceControllers;
ITraceController* m_pActiveController; // The controller currently driving the TraceSession.
// We can have only one active controller at a time.
TraceSessionState m_sessionState; // Current state of the TraceSession
rdfChunkFileWriter* m_pChunkFileWriter; // Helper struct that manages create chunk file streams
// and write data chunks
rdfStream* m_pCurrentStream; // Active RDF stream for writing chunks
Pal::int32 m_currentChunkIndex; // The current chunk index of the RDF stream
bool m_tracingEnabled; // Flag indicating UberTrace tracing is enabled tool-side
void* m_pConfigData; // Buffer containing the cached trace configurationn
size_t m_configDataSize; // Size of the cached trace config buffer
bool m_cancelingTrace; // Indicates that a cancel signal has been received and trace cancelation
// is in progress.
};
} // GpuUtil