CI: Relax timestamp checking (#1189)

* Relax timestamp checking

- Prevent recurring CI failures that have no remedy until HSA/driver issues are resolved

* Replace "cc" abbreviation in tests with "counter-collection"

* Update CODEOWNERS to explicitly include jrmadsen for source/include

* Extra logging in rocprofiler tool library

* Tweak aborted-app test

- remove counter collection as part of the test
Tá an tiomantas seo le fáil i:
Jonathan R. Madsen
2024-11-06 23:32:47 -06:00
tiomanta ag GitHub
tuismitheoir 6564419357
tiomantas 98858b60ec
D'athraigh 12 comhad le 127 breiseanna agus 87 scriosta
+12 -6
Féach ar an gComhad
@@ -342,14 +342,20 @@ async_copy_handler(hsa_signal_value_t signal_value, void* arg)
{
_profile_time = tracing::adjust_profiling_time(
"memcpy",
"hsa_amd_profiling_get_async_copy_time",
_profile_time,
tracing::profiling_time{HSA_STATUS_SUCCESS, _data->start_ts, ts});
// if we encounter this in CI, it will cause test to fail
ROCP_CI_LOG_IF(ERROR, _profile_time.end < _profile_time.start)
<< "hsa_amd_profiling_get_async_copy_time for returned async times where the end time ("
<< _profile_time.end << ") was less than the start time (" << _profile_time.start
<< ")";
}
else
{
ROCP_CI_LOG(ERROR) << fmt::format(
"hsa_amd_profiling_get_async_copy_time for the {} copy operation from agent-{} to "
"agent-{} returned status={} :: {}",
std::string_view{hsa::async_copy::name_by_id(_data->direction)},
CHECK_NOTNULL(agent::get_agent(_data->src_agent))->node_id,
CHECK_NOTNULL(agent::get_agent(_data->dst_agent))->node_id,
static_cast<int>(copy_time_status),
hsa::get_hsa_status_string(copy_time_status));
}
// get the contexts that were active when the signal was created
@@ -55,27 +55,22 @@ get_dispatch_time(hsa_agent_t _hsa_agent,
if(_profile_time.status == HSA_STATUS_SUCCESS)
{
// if we encounter this in CI, it will cause test to fail
ROCP_CI_LOG_IF(ERROR, _profile_time.end < _profile_time.start)
<< "hsa_amd_profiling_get_dispatch_time for kernel_id=" << _kernel_id
<< " on rocprofiler_agent="
<< CHECK_NOTNULL(agent::get_rocprofiler_agent(_hsa_agent))->node_id
<< " returned dispatch times where the end time (" << _profile_time.end
<< ") was less than the start time (" << _profile_time.start << ")";
_profile_time = tracing::adjust_profiling_time(
"dispatch",
"hsa_amd_profiling_get_dispatch_time",
_profile_time,
tracing::profiling_time{
HSA_STATUS_SUCCESS, _baseline.value_or(dispatch_time.start), ts});
}
else
{
ROCP_CI_LOG(ERROR) << "hsa_amd_profiling_get_dispatch_time for kernel id=" << _kernel_id
<< " on rocprofiler_agent="
<< CHECK_NOTNULL(agent::get_rocprofiler_agent(_hsa_agent))->id.handle
<< " returned status=" << dispatch_time_status
<< " :: " << hsa::get_hsa_status_string(dispatch_time_status);
ROCP_CI_LOG(ERROR) << fmt::format(
"hsa_amd_profiling_get_dispatch_time for kernel id={} on agent-{} returned status={} "
":: {}",
_kernel_id,
CHECK_NOTNULL(agent::get_rocprofiler_agent(_hsa_agent))->node_id,
static_cast<int>(dispatch_time_status),
hsa::get_hsa_status_string(dispatch_time_status));
}
return _profile_time;
+31 -10
Féach ar an gComhad
@@ -72,7 +72,10 @@ struct profiling_time
};
inline profiling_time
adjust_profiling_time(std::string_view _label, profiling_time _value, profiling_time&& _bounds)
adjust_profiling_time(std::string_view _label,
std::string_view _responsible,
profiling_time _value,
profiling_time&& _bounds)
{
static auto sysclock_period = hsa::get_hsa_timestamp_period();
static auto normalize_env = common::get_env("ROCPROFILER_CI_FREQ_SCALE_TIMESTAMPS", false);
@@ -84,19 +87,21 @@ adjust_profiling_time(std::string_view _label, profiling_time _value, profiling_
if(strict_ts_env)
{
ROCP_FATAL_IF(ROCPROFILER_UNLIKELY(_value.end < _value.start))
<< fmt::format("Invalid {} time value: {} end time ({}) is less than the {} start time "
"({}) :: difference={}",
ROCP_FATAL_IF(ROCPROFILER_UNLIKELY(_value.start > _value.end))
<< fmt::format("{} returned invalid {} time value: {} start time is greater than the "
"{} end time ({} > {}) :: difference={}",
_responsible,
_label,
_label,
_value.end,
_label,
_value.start,
(_value.end - _value.start));
_value.end,
(_value.start - _value.end));
ROCP_FATAL_IF(ROCPROFILER_UNLIKELY(_value.start < _bounds.start))
<< fmt::format("Invalid {} time value: {} start time ({}) is less than the enqueue "
"time on the CPU ({}) :: difference={}",
<< fmt::format("{} returned invalid {} time value: {} start time is before the API "
"call enqueuing the operation on the CPU ({} < {}) :: difference={}",
_responsible,
_label,
_label,
_value.start,
@@ -105,8 +110,9 @@ adjust_profiling_time(std::string_view _label, profiling_time _value, profiling_
(_bounds.start - _value.start));
ROCP_FATAL_IF(ROCPROFILER_UNLIKELY(_value.end > _bounds.end))
<< fmt::format("Invalid {} time value: {} end time ({}) is greater than the current "
"time on the CPU ({}) :: difference={}",
<< fmt::format("{} returned invalid {} time value: {} end time is greater than the "
"current time on the CPU ({} > {}) :: difference={}",
_responsible,
_label,
_label,
_value.end,
@@ -115,6 +121,21 @@ adjust_profiling_time(std::string_view _label, profiling_time _value, profiling_
(_value.end - _bounds.end));
}
if(_value.start > _value.end)
{
ROCP_ERROR << fmt::format(
"{} returned {} times where the start time is after end time ({} > {}) :: "
"difference={}. Swapping the values. Set the environment variable "
"ROCPROFILER_CI_STRICT_TIMESTAMPS=1 to cause a failure instead",
_responsible,
_label,
_value.start,
_value.end,
(_value.start - _value.end));
std::swap(_value.start, _value.end);
}
// below are hacks for clock skew issues:
//
// the timestamp of this handler will always be after when the profiling time ended