P4 to Git Change 1507569 by gandryey@gera-w8 on 2018/01/24 17:56:10
SWDEV-142271 - Performance drop is observed in Ocean Surface Simulation of Compubenchcl in 17.50 when compared to 17.Q4.1
- Rewrite the adaptive mode for waveliimiter. Make sure the performance feedback corresponds to the right wave count. Add the new sampling logic to find the best number, based on average performance.
Affected files ...
... //depot/stg/opencl/drivers/opencl/runtime/device/device.hpp#295 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuwavelimiter.cpp#14 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/gpu/gpuwavelimiter.hpp#10 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palkernel.hpp#15 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.cpp#71 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palvirtual.hpp#39 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palwavelimiter.cpp#4 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palwavelimiter.hpp#5 edit
... //depot/stg/opencl/drivers/opencl/runtime/platform/command.cpp#80 edit
... //depot/stg/opencl/drivers/opencl/runtime/platform/command.hpp#88 edit
... //depot/stg/opencl/drivers/opencl/runtime/utils/flags.hpp#282 edit
[ROCm/clr commit: 1e90e581d6]
이 커밋은 다음에 포함됨:
@@ -75,7 +75,7 @@ class Options;
|
||||
} // option
|
||||
|
||||
struct ProfilingCallback : public amd::HeapObject {
|
||||
virtual void callback(ulong duration) = 0;
|
||||
virtual void callback(ulong duration, uint32_t waves) = 0;
|
||||
};
|
||||
}
|
||||
|
||||
@@ -970,6 +970,10 @@ class Kernel : public amd::HeapObject {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
virtual uint getWavesPerSH(const device::VirtualDevice* vdv) const {
|
||||
return 0;
|
||||
}
|
||||
|
||||
void setVecTypeHint(const std::string& hint) { workGroupInfo_.compileVecTypeHint_ = hint; }
|
||||
|
||||
void setLocalMemSize(size_t size) { workGroupInfo_.localMemSize_ = size; }
|
||||
|
||||
@@ -121,7 +121,7 @@ void WLAlgorithmSmooth::outputTrace() {
|
||||
}
|
||||
|
||||
|
||||
void WLAlgorithmSmooth::callback(ulong duration) {
|
||||
void WLAlgorithmSmooth::callback(ulong duration, uint32_t waves) {
|
||||
dumper_.addData(duration, currWaves_, static_cast<char>(state_));
|
||||
|
||||
if (!enable_ || (duration == 0)) {
|
||||
@@ -205,65 +205,6 @@ void WaveLimiter::DataDumper::addData(ulong time, uint wave, char state) {
|
||||
state_.push_back(state);
|
||||
}
|
||||
|
||||
WLAlgorithmAvrg::WLAlgorithmAvrg(WaveLimiterManager* manager, uint seqNum, bool enable,
|
||||
bool enableDump)
|
||||
: WaveLimiter(manager, seqNum, enable, enableDump) {
|
||||
measure_.resize(MaxWave + 1);
|
||||
clear(measure_);
|
||||
countAll_ = 0;
|
||||
}
|
||||
|
||||
WLAlgorithmAvrg::~WLAlgorithmAvrg() {}
|
||||
|
||||
void WLAlgorithmAvrg::outputTrace() {
|
||||
if (!traceStream_.is_open()) {
|
||||
return;
|
||||
}
|
||||
|
||||
traceStream_ << "[WaveLimiter] " << manager_->name() << " state=" << state_
|
||||
<< " currWaves=" << currWaves_ << " waves=" << waves_ << " bestWave=" << bestWave_
|
||||
<< '\n';
|
||||
output(traceStream_, "\n measure = ", measure_);
|
||||
traceStream_ << "\n\n";
|
||||
}
|
||||
|
||||
|
||||
void WLAlgorithmAvrg::callback(ulong duration) {
|
||||
dumper_.addData(duration, currWaves_, static_cast<char>(state_));
|
||||
|
||||
if (!enable_) {
|
||||
return;
|
||||
}
|
||||
|
||||
countAll_++;
|
||||
|
||||
switch (state_) {
|
||||
case WARMUP:
|
||||
state_ = ADAPT;
|
||||
case ADAPT:
|
||||
measure_[waves_] += duration;
|
||||
if (countAll_ <= MaxWave * 5) {
|
||||
waves_--;
|
||||
if (waves_ == 0) {
|
||||
waves_ = MaxWave;
|
||||
}
|
||||
} else {
|
||||
bestWave_ = MaxWave;
|
||||
for (uint i = 1; i < MaxWave; i++) {
|
||||
if (measure_[i] < measure_[bestWave_]) {
|
||||
bestWave_ = i;
|
||||
}
|
||||
}
|
||||
waves_ = bestWave_;
|
||||
state_ = RUN;
|
||||
}
|
||||
break;
|
||||
case RUN:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
WaveLimiterManager::WaveLimiterManager(device::Kernel* kernel, const uint simdPerSH)
|
||||
: owner_(kernel), enable_(false), enableDump_(!flagIsDefault(GPU_WAVE_LIMIT_DUMP)) {
|
||||
setIfNotDefault(simdPerSH_, GPU_WAVE_LIMIT_CU_PER_SH, simdPerSH);
|
||||
|
||||
@@ -66,7 +66,7 @@ class WaveLimiter : public amd::ProfilingCallback {
|
||||
static uint RunCount; // Number of kernel executions for normal run
|
||||
|
||||
//! Call back from Event::recordProfilingInfo to get execution time.
|
||||
virtual void callback(ulong duration) = 0;
|
||||
virtual void callback(ulong duration, uint32_t waves) = 0;
|
||||
|
||||
//! Output trace of measurement/adaptation.
|
||||
virtual void outputTrace() = 0;
|
||||
@@ -109,20 +109,7 @@ class WLAlgorithmSmooth : public WaveLimiter {
|
||||
void clearData();
|
||||
|
||||
//! Call back from Event::recordProfilingInfo to get execution time.
|
||||
void callback(ulong duration);
|
||||
|
||||
//! Output trace of measurement/adaptation.
|
||||
void outputTrace();
|
||||
};
|
||||
|
||||
class WLAlgorithmAvrg : public WaveLimiter {
|
||||
public:
|
||||
explicit WLAlgorithmAvrg(WaveLimiterManager* manager, uint seqNum, bool enable, bool enableDump);
|
||||
virtual ~WLAlgorithmAvrg();
|
||||
|
||||
private:
|
||||
//! Call back from Event::recordProfilingInfo to get execution time.
|
||||
void callback(ulong duration);
|
||||
void callback(ulong duration, uint32_t waves);
|
||||
|
||||
//! Output trace of measurement/adaptation.
|
||||
void outputTrace();
|
||||
|
||||
@@ -200,12 +200,12 @@ class HSAILKernel : public device::Kernel {
|
||||
//! Get profiling callback object
|
||||
virtual amd::ProfilingCallback* getProfilingCallback(const device::VirtualDevice* vdev) {
|
||||
return waveLimiter_.getProfilingCallback(vdev);
|
||||
}
|
||||
};
|
||||
|
||||
//! Get waves per shader array to be used for kernel execution.
|
||||
uint getWavesPerSH(const device::VirtualDevice* vdev) const {
|
||||
virtual uint getWavesPerSH(const device::VirtualDevice* vdev) const {
|
||||
return waveLimiter_.getWavesPerSH(vdev);
|
||||
}
|
||||
};
|
||||
|
||||
private:
|
||||
//! Disable copy constructor
|
||||
|
||||
@@ -204,10 +204,12 @@ void VirtualGPU::Queue::addCmdDoppRef(Pal::IGpuMemory* iMem, bool lastDoppCmd, b
|
||||
palDoppRefs_.push_back(doppRef);
|
||||
}
|
||||
|
||||
template <bool avoidBarrierSubmit>
|
||||
uint VirtualGPU::Queue::submit(bool forceFlush) {
|
||||
cmdCnt_++;
|
||||
uint id = cmdBufIdCurrent_;
|
||||
if ((cmdCnt_ > MaxCommands) || forceFlush) {
|
||||
bool flushCmd = ((cmdCnt_ > MaxCommands) || forceFlush) && !avoidBarrierSubmit;
|
||||
if (flushCmd) {
|
||||
if (!flush()) {
|
||||
return GpuEvent::InvalidID;
|
||||
}
|
||||
@@ -1972,7 +1974,8 @@ bool VirtualGPU::submitKernelInternal(const amd::NDRangeContainer& sizes, const
|
||||
}
|
||||
dispatchParam.pCpuAqlCode = hsaKernel.cpuAqlCode();
|
||||
dispatchParam.hsaQueueVa = hsaQueueMem_->vmAddress();
|
||||
dispatchParam.wavesPerSh = hsaKernel.getWavesPerSH(this);
|
||||
dispatchParam.wavesPerSh = (enqueueEvent != nullptr) ?
|
||||
enqueueEvent->profilingInfo().waves_ : 0;
|
||||
dispatchParam.useAtc = dev().settings().svmFineGrainSystem_ ? true : false;
|
||||
// Run AQL dispatch in HW
|
||||
eventBegin(MainEngine);
|
||||
|
||||
@@ -135,6 +135,7 @@ class VirtualGPU : public device::VirtualDevice {
|
||||
|
||||
//! Flushes the current command buffer to HW
|
||||
//! Returns ID associated with the submission
|
||||
template <bool avoidBarrierSubmit = false>
|
||||
uint submit(bool forceFlush);
|
||||
|
||||
bool flush();
|
||||
@@ -449,9 +450,7 @@ class VirtualGPU : public device::VirtualDevice {
|
||||
barrier.pTransitions = &trans;
|
||||
barrier.waitPoint = Pal::HwPipePreCs;
|
||||
iCmd()->CmdBarrier(barrier);
|
||||
if (!profiling()) {
|
||||
queues_[engineID_]->submit(false);
|
||||
}
|
||||
queues_[engineID_]->submit<true>(false);
|
||||
}
|
||||
|
||||
void eventBegin(EngineType engId) const {
|
||||
|
||||
@@ -12,18 +12,15 @@ using namespace std;
|
||||
namespace pal {
|
||||
|
||||
uint WaveLimiter::MaxWave;
|
||||
uint WaveLimiter::WarmUpCount;
|
||||
uint WaveLimiter::RunCount;
|
||||
uint WLAlgorithmSmooth::AdaptCount;
|
||||
uint WLAlgorithmSmooth::AbandonThresh;
|
||||
uint WLAlgorithmSmooth::DscThresh;
|
||||
|
||||
WaveLimiter::WaveLimiter(WaveLimiterManager* manager, uint seqNum, bool enable, bool enableDump)
|
||||
: manager_(manager), dumper_(manager_->name() + "_" + std::to_string(seqNum), enableDump) {
|
||||
setIfNotDefault(SIMDPerSH_, GPU_WAVE_LIMIT_CU_PER_SH, manager->getSimdPerSH());
|
||||
MaxWave = GPU_WAVE_LIMIT_MAX_WAVE;
|
||||
WarmUpCount = GPU_WAVE_LIMIT_WARMUP;
|
||||
RunCount = GPU_WAVE_LIMIT_RUN * MaxWave;
|
||||
AdaptCount = MaxContinuousSamples * 2 * (MaxWave + 1);
|
||||
|
||||
state_ = WARMUP;
|
||||
if (!flagIsDefault(GPU_WAVE_LIMIT_TRACE)) {
|
||||
@@ -31,9 +28,11 @@ WaveLimiter::WaveLimiter(WaveLimiterManager* manager, uint seqNum, bool enable,
|
||||
}
|
||||
|
||||
waves_ = MaxWave;
|
||||
currWaves_ = MaxWave;
|
||||
bestWave_ = MaxWave;
|
||||
enable_ = enable;
|
||||
enable_ = (SIMDPerSH_ == 0) ? false : enable;
|
||||
bestWave_ = (enable_) ? MaxWave : 0;
|
||||
sampleCount_ = 0;
|
||||
resultCount_ = 0;
|
||||
numContinuousSamples_ = 0;
|
||||
}
|
||||
|
||||
WaveLimiter::~WaveLimiter() {
|
||||
@@ -43,22 +42,28 @@ WaveLimiter::~WaveLimiter() {
|
||||
}
|
||||
|
||||
uint WaveLimiter::getWavesPerSH() {
|
||||
currWaves_ = waves_;
|
||||
// Generate different wave counts in the adaptation mode
|
||||
if ((state_ == ADAPT) && (sampleCount_ < AdaptCount)) {
|
||||
if (numContinuousSamples_ == 0) {
|
||||
waves_ = ++waves_ % (MaxWave + 1);
|
||||
}
|
||||
numContinuousSamples_ = ++numContinuousSamples_ % MaxContinuousSamples;
|
||||
++sampleCount_;
|
||||
}
|
||||
else {
|
||||
waves_ = bestWave_;
|
||||
}
|
||||
return waves_ * SIMDPerSH_;
|
||||
}
|
||||
|
||||
WLAlgorithmSmooth::WLAlgorithmSmooth(WaveLimiterManager* manager, uint seqNum, bool enable,
|
||||
bool enableDump)
|
||||
: WaveLimiter(manager, seqNum, enable, enableDump) {
|
||||
AdaptCount = 2 * MaxWave + 1;
|
||||
AbandonThresh = GPU_WAVE_LIMIT_ABANDON;
|
||||
DscThresh = GPU_WAVE_LIMIT_DSC_THRESH;
|
||||
|
||||
dynRunCount_ = RunCount;
|
||||
measure_.resize(MaxWave + 1);
|
||||
reference_.resize(MaxWave + 1);
|
||||
trial_.resize(MaxWave + 1);
|
||||
ratio_.resize(MaxWave + 1);
|
||||
adpMeasure_.resize(MaxWave + 1);
|
||||
adpSampleCnt_.resize(MaxWave + 1);
|
||||
runMeasure_.resize(MaxWave + 1);
|
||||
runSampleCnt_.resize(MaxWave + 1);
|
||||
|
||||
clearData();
|
||||
}
|
||||
@@ -68,41 +73,13 @@ WLAlgorithmSmooth::~WLAlgorithmSmooth() {}
|
||||
void WLAlgorithmSmooth::clearData() {
|
||||
waves_ = MaxWave;
|
||||
countAll_ = 0;
|
||||
clear(measure_);
|
||||
clear(reference_);
|
||||
clear(trial_);
|
||||
clear(ratio_);
|
||||
discontinuous_ = false;
|
||||
clear(adpMeasure_);
|
||||
clear(adpSampleCnt_);
|
||||
dataCount_ = 0;
|
||||
}
|
||||
|
||||
void WLAlgorithmSmooth::updateData(ulong time) {
|
||||
auto count = dataCount_ - 1;
|
||||
assert(count < 2 * MaxWave + 1);
|
||||
assert(time > 0);
|
||||
assert(currWaves_ == waves_);
|
||||
if (count % 2 == 0) {
|
||||
assert(waves_ == MaxWave);
|
||||
auto pos = count / 2;
|
||||
measure_[pos] = time;
|
||||
if (pos > 0) {
|
||||
auto wave = MaxWave + 1 - pos;
|
||||
if (abs(static_cast<long>(measure_[pos - 1]) - static_cast<long>(measure_[pos])) * 100 /
|
||||
measure_[pos] >
|
||||
DscThresh) {
|
||||
discontinuous_ = true;
|
||||
}
|
||||
reference_[wave] = (time + measure_[pos - 1]) / 2;
|
||||
ratio_[wave] = trial_[wave] * 100 / reference_[wave];
|
||||
if (ratio_[bestWave_] > ratio_[wave] && !discontinuous_) {
|
||||
bestWave_ = wave;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
assert(waves_ == MaxWave - count / 2);
|
||||
trial_[waves_] = time;
|
||||
}
|
||||
outputTrace();
|
||||
|
||||
}
|
||||
|
||||
void WLAlgorithmSmooth::outputTrace() {
|
||||
@@ -111,17 +88,25 @@ void WLAlgorithmSmooth::outputTrace() {
|
||||
}
|
||||
|
||||
traceStream_ << "[WaveLimiter] " << manager_->name() << " state=" << state_
|
||||
<< " currWaves=" << currWaves_ << " waves=" << waves_ << " bestWave=" << bestWave_
|
||||
<< '\n';
|
||||
output(traceStream_, "\n measure = ", measure_);
|
||||
output(traceStream_, "\n reference = ", reference_);
|
||||
output(traceStream_, "\n ratio = ", ratio_);
|
||||
<< " waves=" << waves_ << " bestWave=" << bestWave_ << '\n';
|
||||
output(traceStream_, "\n adaptive measure = ", adpMeasure_);
|
||||
output(traceStream_, "\n adaptive smaple count = ", adpSampleCnt_);
|
||||
output(traceStream_, "\n run measure = ", runMeasure_);
|
||||
output(traceStream_, "\n run smaple count = ", runSampleCnt_);
|
||||
traceStream_ << "\n % time from the previous runs to the best wave: ";
|
||||
float min = static_cast<float>(adpMeasure_[bestWave_]) / adpSampleCnt_[bestWave_];
|
||||
for (uint i = 0; i < (MaxWave + 1); ++i) {
|
||||
runSampleCnt_[i] = (runSampleCnt_[i] == 0) ? 1 : runSampleCnt_[i];
|
||||
float average = static_cast<float>(runMeasure_[i]) / runSampleCnt_[i];
|
||||
traceStream_ << (average * 100 / min) << " ";
|
||||
}
|
||||
traceStream_ << "\n run count = " << dynRunCount_;
|
||||
traceStream_ << "\n\n";
|
||||
}
|
||||
|
||||
|
||||
void WLAlgorithmSmooth::callback(ulong duration) {
|
||||
dumper_.addData(duration, currWaves_, static_cast<char>(state_));
|
||||
void WLAlgorithmSmooth::callback(ulong duration, uint32_t waves) {
|
||||
dumper_.addData(duration, waves, static_cast<char>(state_));
|
||||
|
||||
if (!enable_ || (duration == 0)) {
|
||||
return;
|
||||
@@ -129,46 +114,77 @@ void WLAlgorithmSmooth::callback(ulong duration) {
|
||||
|
||||
countAll_++;
|
||||
|
||||
waves /= SIMDPerSH_;
|
||||
// Collect the time for the current wave count
|
||||
runMeasure_[waves] += duration;
|
||||
runSampleCnt_[waves]++;
|
||||
|
||||
switch (state_) {
|
||||
case WARMUP:
|
||||
if (countAll_ < WarmUpCount) {
|
||||
return;
|
||||
}
|
||||
state_ = ADAPT;
|
||||
bestWave_ = MaxWave;
|
||||
clearData();
|
||||
return;
|
||||
case ADAPT:
|
||||
assert(duration > 0);
|
||||
if (waves_ == currWaves_) {
|
||||
dataCount_++;
|
||||
updateData(duration);
|
||||
waves_ = MaxWave + 1 - dataCount_ / 2;
|
||||
if (dataCount_ == 1 || (dataCount_ < AdaptCount && !discontinuous_ &&
|
||||
(dataCount_ % 2 == 0 || ratio_[waves_] < AbandonThresh))) {
|
||||
if (dataCount_ % 2 == 1) {
|
||||
--waves_;
|
||||
} else {
|
||||
waves_ = MaxWave;
|
||||
// Wave count 0 indicates the satrt of adaptation
|
||||
if ((waves == 0) || (resultCount_ > 0)) {
|
||||
// Scale time to us
|
||||
adpMeasure_[waves] += duration;
|
||||
adpSampleCnt_[waves]++;
|
||||
resultCount_++;
|
||||
// If the end of adaptation is reached, then analyze the results
|
||||
if (resultCount_ == AdaptCount) {
|
||||
// Reset the counters
|
||||
resultCount_ = sampleCount_ = 0;
|
||||
float min = std::numeric_limits<float>::max();
|
||||
uint32_t best = bestWave_;
|
||||
// Check performance for the previous run if it's available
|
||||
if (runSampleCnt_[bestWave_] > 0) {
|
||||
min = static_cast<float>(runMeasure_[bestWave_]) / runSampleCnt_[bestWave_];
|
||||
}
|
||||
return;
|
||||
else if (adpSampleCnt_[MaxWave] > 0) {
|
||||
min = static_cast<float>(adpMeasure_[MaxWave]) / adpSampleCnt_[MaxWave];
|
||||
bestWave_ = MaxWave;
|
||||
}
|
||||
// Find the fastest average time
|
||||
float reference = min;
|
||||
for (uint i = MaxWave; i > 0; --i) {
|
||||
float average;
|
||||
if (adpSampleCnt_[i] > 0) {
|
||||
average = static_cast<float>(adpMeasure_[i]) / adpSampleCnt_[i];
|
||||
}
|
||||
else {
|
||||
average = 0.0f;
|
||||
}
|
||||
if (average < min) {
|
||||
min = average;
|
||||
bestWave_ = i;
|
||||
}
|
||||
}
|
||||
// Check for 5% acceptance
|
||||
if ((min * 1.05f > reference) || (bestWave_ == best)) {
|
||||
bestWave_ = best;
|
||||
// Increase the run time if the same wave count is the best
|
||||
dynRunCount_ += RunCount;
|
||||
dynRunCount_++;
|
||||
}
|
||||
else {
|
||||
dynRunCount_ = RunCount;
|
||||
}
|
||||
state_ = RUN;
|
||||
outputTrace();
|
||||
// Start to collect the new data for the best wave
|
||||
countAll_ = 0;
|
||||
runMeasure_[bestWave_] = 0;
|
||||
runSampleCnt_[bestWave_] = 0;
|
||||
}
|
||||
waves_ = bestWave_;
|
||||
if (dataCount_ >= AdaptCount) {
|
||||
dynRunCount_ = RunCount;
|
||||
} else {
|
||||
dynRunCount_ = AdaptCount;
|
||||
}
|
||||
countAll_ = rand() % MaxWave;
|
||||
state_ = RUN;
|
||||
}
|
||||
return;
|
||||
case WARMUP:
|
||||
case RUN:
|
||||
if (countAll_ < dynRunCount_) {
|
||||
return;
|
||||
}
|
||||
if (state_ == WARMUP) {
|
||||
runSampleCnt_[bestWave_] = 0;
|
||||
}
|
||||
state_ = ADAPT;
|
||||
bestWave_ = MaxWave;
|
||||
clearData();
|
||||
return;
|
||||
}
|
||||
@@ -204,64 +220,6 @@ void WaveLimiter::DataDumper::addData(ulong time, uint wave, char state) {
|
||||
state_.push_back(state);
|
||||
}
|
||||
|
||||
WLAlgorithmAvrg::WLAlgorithmAvrg(WaveLimiterManager* manager, uint seqNum, bool enable,
|
||||
bool enableDump)
|
||||
: WaveLimiter(manager, seqNum, enable, enableDump) {
|
||||
measure_.resize(MaxWave + 1);
|
||||
clear(measure_);
|
||||
countAll_ = 0;
|
||||
}
|
||||
|
||||
WLAlgorithmAvrg::~WLAlgorithmAvrg() {}
|
||||
|
||||
void WLAlgorithmAvrg::outputTrace() {
|
||||
if (!traceStream_.is_open()) {
|
||||
return;
|
||||
}
|
||||
|
||||
traceStream_ << "[WaveLimiter] " << manager_->name() << " state=" << state_
|
||||
<< " currWaves=" << currWaves_ << " waves=" << waves_ << " bestWave=" << bestWave_
|
||||
<< '\n';
|
||||
output(traceStream_, "\n measure = ", measure_);
|
||||
traceStream_ << "\n\n";
|
||||
}
|
||||
|
||||
void WLAlgorithmAvrg::callback(ulong duration) {
|
||||
dumper_.addData(duration, currWaves_, static_cast<char>(state_));
|
||||
|
||||
if (!enable_) {
|
||||
return;
|
||||
}
|
||||
|
||||
countAll_++;
|
||||
|
||||
switch (state_) {
|
||||
case WARMUP:
|
||||
state_ = ADAPT;
|
||||
case ADAPT:
|
||||
measure_[waves_] += duration;
|
||||
if (countAll_ <= MaxWave * 5) {
|
||||
waves_--;
|
||||
if (waves_ == 0) {
|
||||
waves_ = MaxWave;
|
||||
}
|
||||
} else {
|
||||
bestWave_ = MaxWave;
|
||||
for (uint i = 1; i < MaxWave; i++) {
|
||||
if (measure_[i] < measure_[bestWave_]) {
|
||||
bestWave_ = i;
|
||||
}
|
||||
}
|
||||
waves_ = bestWave_;
|
||||
state_ = RUN;
|
||||
}
|
||||
break;
|
||||
case RUN:
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
WaveLimiterManager::WaveLimiterManager(device::Kernel* kernel, const uint simdPerSH)
|
||||
: owner_(kernel), enable_(false), enableDump_(!flagIsDefault(GPU_WAVE_LIMIT_DUMP)) {
|
||||
setIfNotDefault(simdPerSH_, GPU_WAVE_LIMIT_CU_PER_SH, simdPerSH);
|
||||
|
||||
@@ -48,7 +48,6 @@ class WaveLimiter : public amd::ProfilingCallback {
|
||||
std::vector<char> state_;
|
||||
};
|
||||
|
||||
std::vector<uint64_t> measure_;
|
||||
bool enable_;
|
||||
uint SIMDPerSH_; // Number of SIMDs per SH
|
||||
uint waves_; // Waves per SIMD to be set
|
||||
@@ -58,14 +57,17 @@ class WaveLimiter : public amd::ProfilingCallback {
|
||||
WaveLimiterManager* manager_;
|
||||
DataDumper dumper_;
|
||||
std::ofstream traceStream_;
|
||||
uint currWaves_; // Current waves per SIMD
|
||||
uint32_t sampleCount_; //!< The number of samples for adaptive mode
|
||||
uint32_t resultCount_; //!< The number of results for adaptive mode
|
||||
uint32_t numContinuousSamples_; //!< The number of samples with the same wave count
|
||||
|
||||
static uint MaxWave; // Maximum number of waves per SIMD
|
||||
static uint WarmUpCount; // Number of kernel executions for warm up
|
||||
static uint RunCount; // Number of kernel executions for normal run
|
||||
const static uint MaxContinuousSamples = 8;
|
||||
static uint AdaptCount; // Number of kernel executions for adapting
|
||||
|
||||
//! Call back from Event::recordProfilingInfo to get execution time.
|
||||
virtual void callback(ulong duration) = 0;
|
||||
virtual void callback(ulong duration, uint32_t waves) = 0;
|
||||
|
||||
//! Output trace of measurement/adaptation.
|
||||
virtual void outputTrace() = 0;
|
||||
@@ -90,17 +92,13 @@ class WLAlgorithmSmooth : public WaveLimiter {
|
||||
virtual ~WLAlgorithmSmooth();
|
||||
|
||||
private:
|
||||
std::vector<uint64_t> reference_;
|
||||
std::vector<uint64_t> trial_;
|
||||
std::vector<uint64_t> ratio_;
|
||||
bool discontinuous_; // Measured data is discontinuous
|
||||
std::vector<uint64_t> adpMeasure_; //!< Accumulated performance in the adaptation mode
|
||||
std::vector<uint32_t> adpSampleCnt_; //!< The number of samples in the adaptation mode
|
||||
std::vector<uint64_t> runMeasure_; //!< Accumulated performance in the run mode
|
||||
std::vector<uint32_t> runSampleCnt_; //!< The number of samples in the run mode
|
||||
uint dynRunCount_;
|
||||
uint dataCount_;
|
||||
|
||||
static uint AdaptCount; // Number of kernel executions for adapting
|
||||
static uint AbandonThresh; // Threshold to abandon adaptation
|
||||
static uint DscThresh; // Threshold for identifying discontinuities
|
||||
|
||||
//! Update measurement data and optimal waves/simd with execution time.
|
||||
void updateData(ulong time);
|
||||
|
||||
@@ -108,20 +106,7 @@ class WLAlgorithmSmooth : public WaveLimiter {
|
||||
void clearData();
|
||||
|
||||
//! Call back from Event::recordProfilingInfo to get execution time.
|
||||
void callback(ulong duration);
|
||||
|
||||
//! Output trace of measurement/adaptation.
|
||||
void outputTrace();
|
||||
};
|
||||
|
||||
class WLAlgorithmAvrg : public WaveLimiter {
|
||||
public:
|
||||
explicit WLAlgorithmAvrg(WaveLimiterManager* manager, uint seqNum, bool enable, bool enableDump);
|
||||
virtual ~WLAlgorithmAvrg();
|
||||
|
||||
private:
|
||||
//! Call back from Event::recordProfilingInfo to get execution time.
|
||||
void callback(ulong duration);
|
||||
virtual void callback(ulong duration, uint32_t waves) override;
|
||||
|
||||
//! Output trace of measurement/adaptation.
|
||||
void outputTrace();
|
||||
|
||||
@@ -61,7 +61,8 @@ uint64_t Event::recordProfilingInfo(cl_int status, uint64_t timeStamp) {
|
||||
default:
|
||||
profilingInfo_.end_ = timeStamp;
|
||||
if (profilingInfo_.callback_ != NULL) {
|
||||
profilingInfo_.callback_->callback(timeStamp - profilingInfo_.start_);
|
||||
profilingInfo_.callback_->callback(timeStamp - profilingInfo_.start_,
|
||||
profilingInfo_.waves_);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -227,7 +228,8 @@ NDRangeKernelCommand::NDRangeKernelCommand(HostQueue& queue, const EventWaitList
|
||||
parameters_ = kernel.parameters().capture(queue.device());
|
||||
auto& device = queue.device();
|
||||
auto devKernel = const_cast<device::Kernel*>(kernel.getDeviceKernel(device));
|
||||
profilingInfo_.setCallback(devKernel->getProfilingCallback(queue.vdev()));
|
||||
profilingInfo_.setCallback(devKernel->getProfilingCallback(
|
||||
queue.vdev()), devKernel->getWavesPerSH(queue.vdev()));
|
||||
fixme_guarantee(parameters_ != NULL && "out of memory");
|
||||
kernel_.retain();
|
||||
}
|
||||
|
||||
@@ -80,7 +80,7 @@ class Event : public RuntimeObject {
|
||||
static const EventWaitList nullWaitList;
|
||||
|
||||
struct ProfilingInfo {
|
||||
ProfilingInfo(bool enabled = false) : enabled_(enabled) {
|
||||
ProfilingInfo(bool enabled = false) : enabled_(enabled), waves_(0) {
|
||||
if (enabled) {
|
||||
clear();
|
||||
callback_ = NULL;
|
||||
@@ -91,7 +91,8 @@ class Event : public RuntimeObject {
|
||||
uint64_t submitted_;
|
||||
uint64_t start_;
|
||||
uint64_t end_;
|
||||
bool enabled_;
|
||||
bool enabled_; //!< Profiling enabled for the wave limiter
|
||||
uint32_t waves_; //!< The number of waves used in a dispatch
|
||||
ProfilingCallback* callback_;
|
||||
void clear() {
|
||||
queued_ = 0ULL;
|
||||
@@ -99,11 +100,12 @@ class Event : public RuntimeObject {
|
||||
start_ = 0ULL;
|
||||
end_ = 0ULL;
|
||||
}
|
||||
void setCallback(ProfilingCallback* callback) {
|
||||
void setCallback(ProfilingCallback* callback, uint32_t waves) {
|
||||
if (callback == NULL) {
|
||||
return;
|
||||
}
|
||||
enabled_ = true;
|
||||
waves_ = waves;
|
||||
clear();
|
||||
callback_ = callback;
|
||||
}
|
||||
|
||||
@@ -193,8 +193,6 @@ release_on_stg(uint, GPU_WAVE_LIMIT_MAX_WAVE, 10, \
|
||||
"Set maximum waves per SIMD to try for wave limiter") \
|
||||
release_on_stg(uint, GPU_WAVE_LIMIT_WARMUP, 100, \
|
||||
"Set warming up kernel execution count for wave limiter") \
|
||||
release_on_stg(uint, GPU_WAVE_LIMIT_ADAPT, 1, \
|
||||
"Set adapting factor for wave limiter") \
|
||||
release_on_stg(uint, GPU_WAVE_LIMIT_RUN, 20, \
|
||||
"Set running factor for wave limiter") \
|
||||
release_on_stg(uint, GPU_WAVE_LIMIT_ABANDON, 105, \
|
||||
|
||||
새 이슈에서 참조
사용자 차단