SWDEV-292018 - Switch to internal signals for markers
Add ref counting to ProfilingSignal class to track the last release. If a signal was used in the marker, then don't reuse it, but create a new one for internal usage. Don't rely on HSA callback for the command status update if there are no pending dispatches. Change-Id: I19f14ed9d80acfe79993b343b2187635f8428a20
此提交包含在:
@@ -319,10 +319,7 @@ void VirtualGPU::MemoryDependency::clear(bool all) {
|
||||
// ================================================================================================
|
||||
VirtualGPU::HwQueueTracker::~HwQueueTracker() {
|
||||
for (auto& signal: signal_list_) {
|
||||
if (signal->signal_.handle != 0) {
|
||||
hsa_signal_destroy(signal->signal_);
|
||||
}
|
||||
delete signal;
|
||||
signal->release();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -374,6 +371,26 @@ hsa_signal_t VirtualGPU::HwQueueTracker::ActiveSignal(
|
||||
// a GPU waiter(which may be not triggered yet) and CPU signal reset below
|
||||
WaitNext();
|
||||
|
||||
if (signal_list_[current_id_]->referenceCount() > 1) {
|
||||
// The signal was assigned to the global marker's event, hence runtime can't reuse it
|
||||
// and needs a new signal
|
||||
std::unique_ptr<ProfilingSignal> signal(new ProfilingSignal());
|
||||
if (signal != nullptr) {
|
||||
hsa_agent_t agent = gpu_.gpu_device();
|
||||
const Settings& settings = gpu_.dev().settings();
|
||||
hsa_agent_t* agents = (settings.system_scope_signal_) ? nullptr : &agent;
|
||||
uint32_t num_agents = (settings.system_scope_signal_) ? 0 : 1;
|
||||
|
||||
if (HSA_STATUS_SUCCESS == hsa_signal_create(0, num_agents, agents, &signal->signal_)) {
|
||||
signal_list_[current_id_]->release();
|
||||
signal_list_[current_id_] = signal.release();
|
||||
} else {
|
||||
assert(!"ProfilingSignal reallocaiton failed! Marker has a conflict with signal reuse!");
|
||||
}
|
||||
} else {
|
||||
assert(!"ProfilingSignal reallocaiton failed! Marker has a conflict with signal reuse!");
|
||||
}
|
||||
}
|
||||
ProfilingSignal* prof_signal = signal_list_[current_id_];
|
||||
// Reset the signal and return
|
||||
hsa_signal_silent_store_relaxed(prof_signal->signal_, init_val);
|
||||
@@ -387,7 +404,23 @@ hsa_signal_t VirtualGPU::HwQueueTracker::ActiveSignal(
|
||||
// If direct dispatch is enabled and the batch head isn't null, then it's a marker and
|
||||
// requires the batch update upon HSA signal completion
|
||||
if (AMD_DIRECT_DISPATCH && (ts->command().GetBatchHead() != nullptr)) {
|
||||
assert(false && "Runtime should not have batch command in ActiveSignal!");
|
||||
uint32_t init_value = kInitSignalValueOne;
|
||||
// If API callback is enabled, then use a blocking signal for AQL queue.
|
||||
// HSA signal will be acquired in SW and released after HSA signal callback
|
||||
if (ts->command().Callback() != nullptr) {
|
||||
ts->SetCallbackSignal(prof_signal->signal_);
|
||||
// Blocks AQL queue from further processing
|
||||
hsa_signal_add_relaxed(prof_signal->signal_, 1);
|
||||
init_value += 1;
|
||||
}
|
||||
hsa_status_t result = hsa_amd_signal_async_handler(prof_signal->signal_,
|
||||
HSA_SIGNAL_CONDITION_LT, init_value, &HsaAmdSignalHandler, ts);
|
||||
if (HSA_STATUS_SUCCESS != result) {
|
||||
LogError("hsa_amd_signal_async_handler() failed to set the handler!");
|
||||
} else {
|
||||
ClPrint(amd::LOG_INFO, amd::LOG_SIG, "Set Handler: handle(0x%lx), timestamp(%p)",
|
||||
prof_signal->signal_.handle, prof_signal);
|
||||
}
|
||||
}
|
||||
if (!sdma_profiling_) {
|
||||
hsa_amd_profiling_async_copy_enable(true);
|
||||
@@ -872,8 +905,7 @@ bool VirtualGPU::dispatchCounterAqlPacket(hsa_ext_amd_aql_pm4_packet_t* packet,
|
||||
}
|
||||
|
||||
// ================================================================================================
|
||||
void VirtualGPU::dispatchBarrierPacket(uint16_t packetHeader,
|
||||
bool skipSignal, const ProfilingSignal* global_signal) {
|
||||
void VirtualGPU::dispatchBarrierPacket(uint16_t packetHeader, bool skipSignal) {
|
||||
const uint32_t queueSize = gpu_queue_->size;
|
||||
const uint32_t queueMask = queueSize - 1;
|
||||
|
||||
@@ -896,16 +928,12 @@ void VirtualGPU::dispatchBarrierPacket(uint16_t packetHeader,
|
||||
barrier_packet_.completion_signal.handle = 0;
|
||||
|
||||
if (!skipSignal) {
|
||||
if (global_signal != nullptr) {
|
||||
barrier_packet_.completion_signal = global_signal->signal_;
|
||||
} else {
|
||||
// Pool size must grow to the size of pending AQL packets
|
||||
const uint32_t pool_size = index - read;
|
||||
// Pool size must grow to the size of pending AQL packets
|
||||
const uint32_t pool_size = index - read;
|
||||
|
||||
// Get active signal for current dispatch if profiling is necessary
|
||||
barrier_packet_.completion_signal =
|
||||
Barriers().ActiveSignal(kInitSignalValueOne, timestamp_, pool_size);
|
||||
}
|
||||
// Get active signal for current dispatch if profiling is necessary
|
||||
barrier_packet_.completion_signal =
|
||||
Barriers().ActiveSignal(kInitSignalValueOne, timestamp_, pool_size);
|
||||
}
|
||||
|
||||
while ((index - hsa_queue_load_read_index_scacquire(gpu_queue_)) >= queueMask);
|
||||
@@ -1226,6 +1254,12 @@ void VirtualGPU::profilingEnd(amd::Command& command) {
|
||||
}
|
||||
command.setData(timestamp_);
|
||||
|
||||
// Update HW event only for batches
|
||||
if ((AMD_DIRECT_DISPATCH) && (command.GetBatchHead() != nullptr)) {
|
||||
timestamp_->Signals().back()->retain();
|
||||
command.SetHwEvent(timestamp_->Signals().back());
|
||||
}
|
||||
|
||||
timestamp_ = nullptr;
|
||||
}
|
||||
}
|
||||
@@ -2889,7 +2923,7 @@ void VirtualGPU::submitKernel(amd::NDRangeKernelCommand& vcmd) {
|
||||
|
||||
queue->profilingEnd(vcmd);
|
||||
} else {
|
||||
// Make sure VirtualGPU has an exclusive access to the resources
|
||||
// Make sure VirtualGPU has an exclusive access to the resources
|
||||
amd::ScopedLock lock(execution());
|
||||
|
||||
profilingBegin(vcmd);
|
||||
@@ -2913,47 +2947,23 @@ void VirtualGPU::submitNativeFn(amd::NativeFnCommand& cmd) {
|
||||
// ================================================================================================
|
||||
void VirtualGPU::submitMarker(amd::Marker& vcmd) {
|
||||
if (AMD_DIRECT_DISPATCH || vcmd.profilingInfo().marker_ts_) {
|
||||
profilingBegin(vcmd);
|
||||
if (timestamp_ != nullptr) {
|
||||
ProfilingSignal* prof_signal = nullptr;
|
||||
// If direct dispatch is enabled and the batch head isn't null, then it's a marker and
|
||||
// requires the batch update upon HSA signal completion
|
||||
if (AMD_DIRECT_DISPATCH) {
|
||||
assert(vcmd.GetBatchHead() != nullptr && "Marker doesn't have batch!");
|
||||
// Make sure VirtualGPU has an exclusive access to the resources
|
||||
amd::ScopedLock lock(execution());
|
||||
if (vcmd.CpuWaitRequested() && hasPendingDispatch_ == false) {
|
||||
// It should be safe to call flush directly if there are not pending dispatches without
|
||||
// HSA signal callback
|
||||
flush(vcmd.GetBatchHead());
|
||||
} else {
|
||||
profilingBegin(vcmd);
|
||||
if (timestamp_ != nullptr) {
|
||||
// Submit a barrier with a cache flushes.
|
||||
dispatchBarrierPacket(kBarrierPacketHeader, false);
|
||||
|
||||
prof_signal = dev().GetGlobalSignal(timestamp_);
|
||||
prof_signal->done_ = false;
|
||||
|
||||
assert(prof_signal != nullptr && "Failed to allocate the global HSA signal!");
|
||||
uint32_t init_value = kInitSignalValueOne;
|
||||
// If API callback is enabled, then use a blocking signal for AQL queue.
|
||||
// HSA signal will be acquired in SW and released after HSA signal callback
|
||||
if (vcmd.Callback() != nullptr) {
|
||||
timestamp_->SetCallbackSignal(prof_signal->signal_);
|
||||
// Blocks AQL queue from further processing
|
||||
hsa_signal_add_relaxed(prof_signal->signal_, 1);
|
||||
init_value += 1;
|
||||
}
|
||||
|
||||
hsa_status_t result = hsa_amd_signal_async_handler(prof_signal->signal_,
|
||||
HSA_SIGNAL_CONDITION_LT, init_value, &HsaAmdSignalHandler, timestamp_);
|
||||
if (HSA_STATUS_SUCCESS != result) {
|
||||
LogError("hsa_amd_signal_async_handler() failed to set the handler!");
|
||||
} else {
|
||||
ClPrint(amd::LOG_INFO, amd::LOG_SIG, "Set Handler: handle(0x%lx), timestamp(%p)",
|
||||
prof_signal->signal_.handle, prof_signal);
|
||||
}
|
||||
// Update HW event only for batches
|
||||
vcmd.SetHwEvent(timestamp_->Signals().back());
|
||||
hasPendingDispatch_ = false;
|
||||
}
|
||||
// Submit a barrier with a cache flushes.
|
||||
dispatchBarrierPacket(kBarrierPacketHeader, false, prof_signal);
|
||||
|
||||
// Don't reset the flag for direct dispatch, because the global signals are out of scope
|
||||
// for internal barrier tracking and SDMA could lose a wait for compute
|
||||
hasPendingDispatch_ = AMD_DIRECT_DISPATCH;
|
||||
profilingEnd(vcmd);
|
||||
}
|
||||
profilingEnd(vcmd);
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
新增問題並參考
封鎖使用者