P4 to Git Change 1510104 by gandryey@gera-w8 on 2018/01/30 16:21:25

SWDEV-144231 - [CQE OCL][DTB][Perf][DTB-BLOCKER] 10% Performance drop observed while running Subtests of IndigoBench due to faulty CL#1507569
	-Wavelimiter has more negative impact than improvements in a few benchmarks.
	- Reduce the number of adaptive samples during the search to 4 per each wave count.
	- Add worstWave_ to skip sampling with low performance settings.

Affected files ...

... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palwavelimiter.cpp#7 edit
... //depot/stg/opencl/drivers/opencl/runtime/device/pal/palwavelimiter.hpp#7 edit
这个提交包含在:
foreman
2018-01-30 16:49:46 -05:00
父节点 a7ca88a40e
当前提交 2e6d7d965a
修改 2 个文件,包含 32 行新增8 行删除
+23 -3
查看文件
@@ -30,6 +30,7 @@ WaveLimiter::WaveLimiter(WaveLimiterManager* manager, uint seqNum, bool enable,
waves_ = MaxWave;
enable_ = (SIMDPerSH_ == 0) ? false : enable;
bestWave_ = (enable_) ? MaxWave : 0;
worstWave_ = 0;
sampleCount_ = 0;
resultCount_ = 0;
numContinuousSamples_ = 0;
@@ -47,6 +48,13 @@ uint WaveLimiter::getWavesPerSH() {
if (numContinuousSamples_ == 0) {
++waves_;
waves_ %= MaxWave + 1;
// Don't execute the wave count with the worst performance
if (waves_ != 0) {
while (worstWave_ >= waves_) {
++waves_;
waves_ %= MaxWave + 1;
}
}
}
++numContinuousSamples_;
numContinuousSamples_ %= MaxContinuousSamples;
@@ -89,8 +97,8 @@ void WLAlgorithmSmooth::outputTrace() {
return;
}
traceStream_ << "[WaveLimiter] " << manager_->name() << " state=" << state_
<< " waves=" << waves_ << " bestWave=" << bestWave_ << '\n';
traceStream_ << "[WaveLimiter] " << manager_->name() << " state=" << state_ <<
" waves=" << waves_ << " bestWave=" << bestWave_ << " worstWave=" << worstWave_ << '\n';
output(traceStream_, "\n adaptive measure = ", adpMeasure_);
output(traceStream_, "\n adaptive smaple count = ", adpSampleCnt_);
output(traceStream_, "\n run measure = ", runMeasure_);
@@ -135,6 +143,7 @@ void WLAlgorithmSmooth::callback(ulong duration, uint32_t waves) {
// Reset the counters
resultCount_ = sampleCount_ = 0;
float min = std::numeric_limits<float>::max();
float max = std::numeric_limits<float>::min();
uint32_t best = bestWave_;
// Check performance for the previous run if it's available
if (runSampleCnt_[bestWave_] > 0) {
@@ -154,10 +163,15 @@ void WLAlgorithmSmooth::callback(ulong duration, uint32_t waves) {
else {
average = 0.0f;
}
if (average < min) {
// More waves have 5% advantage over the lower number
if (average * 1.05f < min) {
min = average;
bestWave_ = i;
}
if (average > max) {
max = average;
worstWave_ = i;
}
}
// Check for 5% acceptance
if ((min * 1.05f > reference) || (bestWave_ == best)) {
@@ -169,6 +183,12 @@ void WLAlgorithmSmooth::callback(ulong duration, uint32_t waves) {
else {
dynRunCount_ = RunCount;
}
// Find the middle between the best and the worst
if (worstWave_ < bestWave_) {
worstWave_ += ((bestWave_ - worstWave_) >> 1);
} else {
worstWave_ = 0;
}
state_ = RUN;
outputTrace();
// Start to collect the new data for the best wave
+9 -5
查看文件
@@ -52,6 +52,7 @@ class WaveLimiter : public amd::ProfilingCallback {
uint SIMDPerSH_; // Number of SIMDs per SH
uint waves_; // Waves per SIMD to be set
uint bestWave_; // Optimal waves per SIMD
uint worstWave_; // Wave number with the worst performance
uint countAll_; // Number of kernel executions
StateKind state_;
WaveLimiterManager* manager_;
@@ -64,7 +65,7 @@ class WaveLimiter : public amd::ProfilingCallback {
static uint MaxWave; // Maximum number of waves per SIMD
static uint RunCount; // Number of kernel executions for normal run
static uint AdaptCount; // Number of kernel executions for adapting
const static uint MaxContinuousSamples = 8;
const static uint MaxContinuousSamples = 2;
//! Call back from Event::recordProfilingInfo to get execution time.
virtual void callback(ulong duration, uint32_t waves) = 0;
@@ -73,8 +74,12 @@ class WaveLimiter : public amd::ProfilingCallback {
virtual void outputTrace() = 0;
template <class T> void clear(T& A) {
uint idx = 0;
for (auto& I : A) {
I = 0;
if (idx > worstWave_) {
I = 0;
}
++idx;
}
}
template <class T> void output(std::ofstream& ofs, const std::string& prompt, T& A) {
@@ -136,9 +141,8 @@ class WaveLimiterManager {
private:
device::Kernel* owner_; // The kernel which owns this object
uint simdPerSH_; // Simd Per SH
std::unordered_map<const device::VirtualDevice*,
WaveLimiter*>
limiters_; // Maps virtual device to wave limiter
std::unordered_map<const device::VirtualDevice*, WaveLimiter*>
limiters_; // Maps virtual device to wave limiter
bool enable_; // Whether the adaptation is enabled
bool enableDump_; // Whether the data dumper is enabled
uint fixed_; // The fixed waves/simd value if not zero