removed gfx940 and gfx941 (#1606)
* removed gfx940 and gfx941
* removed gfx940 and gfx941
* Update "gfx94" to "gfx942" in init.cc
* Updated remaining "gfx94" updates to "gfx942"
* Update filenames and variables from gfx940 to gfx942
---------
Co-authored-by: akolliasAMD <akollias@amd.com>
[ROCm/rccl commit: 6505639cf4]
This commit is contained in:
zatwierdzone przez
GitHub
rodzic
a95b2c9fb7
commit
e95578ef4c
@@ -169,7 +169,7 @@ static gdr_t ncclGdrInit() {
|
||||
return NULL;
|
||||
}
|
||||
GcnArchNameFormat(devProp.gcnArchName, gcnArchNameSubstr);
|
||||
if (IsArchMatch(gcnArchNameSubstr, "gfx94")) {
|
||||
if (IsArchMatch(gcnArchNameSubstr, "gfx942")) {
|
||||
INFO(NCCL_INIT, "Enabled GDRCopy equivalent memory allocation on %s", gcnArchNameSubstr);
|
||||
return (gdr_t)0x12345678L;
|
||||
} else {
|
||||
|
||||
@@ -344,7 +344,7 @@ struct rccl_float8
|
||||
// default constructor
|
||||
HIP_HOST_DEVICE rccl_float8() = default;
|
||||
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
// device specific optimized F8 down-conversion code
|
||||
|
||||
template <bool stochastic_rounding = false>
|
||||
@@ -381,10 +381,10 @@ struct rccl_float8
|
||||
return i8data;
|
||||
}
|
||||
|
||||
#endif // __gfx940__
|
||||
#endif // __gfx942__
|
||||
|
||||
// constructor from float
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
|
||||
// NOTE: ON-DEVICE... always optimal bias
|
||||
explicit HIP_DEVICE rccl_float8(float v,
|
||||
@@ -402,7 +402,7 @@ struct rccl_float8
|
||||
// Host only implementation using s/w simulation
|
||||
explicit HIP_HOST
|
||||
#else
|
||||
// both Host and DEVICE for non-gfx940 using s/w simulation
|
||||
// both Host and DEVICE for non-gfx942 using s/w simulation
|
||||
explicit HIP_HOST_DEVICE
|
||||
#endif
|
||||
rccl_float8(float v,
|
||||
@@ -446,7 +446,7 @@ struct rccl_float8
|
||||
}
|
||||
|
||||
// convert to float
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
// upcast using device specific intrinsic
|
||||
explicit inline HIP_DEVICE operator float() const
|
||||
{
|
||||
@@ -460,7 +460,7 @@ struct rccl_float8
|
||||
}
|
||||
|
||||
explicit inline HIP_HOST operator float() const
|
||||
#else // non gfx940
|
||||
#else // non gfx942
|
||||
explicit inline HIP_HOST_DEVICE operator float() const
|
||||
#endif
|
||||
{
|
||||
@@ -511,7 +511,7 @@ struct rccl_bfloat8
|
||||
// default constructor
|
||||
HIP_HOST_DEVICE rccl_bfloat8() = default;
|
||||
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
// device specific optimized F8 down-conversion code
|
||||
|
||||
template <bool stochastic_rounding = false>
|
||||
@@ -548,10 +548,10 @@ struct rccl_bfloat8
|
||||
return i8data;
|
||||
}
|
||||
|
||||
#endif // __gfx940__
|
||||
#endif // __gfx942__
|
||||
|
||||
// constructor from float
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
|
||||
// NOTE: ON-DEVICE... always optimal bias
|
||||
explicit HIP_DEVICE rccl_bfloat8(float v,
|
||||
@@ -569,7 +569,7 @@ struct rccl_bfloat8
|
||||
// Host only implementation using s/w simulation
|
||||
explicit HIP_HOST
|
||||
#else
|
||||
// both Host and DEVICE for non-gfx940 using s/w simulation
|
||||
// both Host and DEVICE for non-gfx942 using s/w simulation
|
||||
explicit HIP_HOST_DEVICE
|
||||
#endif
|
||||
rccl_bfloat8(float v,
|
||||
@@ -613,7 +613,7 @@ struct rccl_bfloat8
|
||||
}
|
||||
|
||||
// convert to float
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
// upcast using device specific intrinsic
|
||||
explicit inline HIP_DEVICE operator float() const
|
||||
{
|
||||
@@ -627,7 +627,7 @@ struct rccl_bfloat8
|
||||
}
|
||||
|
||||
explicit inline HIP_HOST operator float() const
|
||||
#else // non gfx940
|
||||
#else // non gfx942
|
||||
explicit inline HIP_HOST_DEVICE operator float() const
|
||||
#endif
|
||||
{
|
||||
@@ -969,7 +969,7 @@ inline __host__ __device__ T explicit_downcast(Ta a, uint32_t rng = 0)
|
||||
return a;
|
||||
}
|
||||
|
||||
// Use h/w intrinsic and optimized version when __gfx940__
|
||||
// Use h/w intrinsic and optimized version when __gfx942__
|
||||
template <
|
||||
typename T,
|
||||
typename Ta,
|
||||
@@ -980,7 +980,7 @@ template <
|
||||
= 0>
|
||||
inline __host__ __device__ T explicit_downcast(Ta a, uint32_t rng)
|
||||
{
|
||||
#if defined(__gfx940__) || defined(__gfx941__) || defined(__gfx942__) || defined(__gfx950__)
|
||||
#if defined(__gfx942__) || defined(__gfx950__)
|
||||
// NOTE: we are directly calling cast_to_f8_from_f32 instead of constructor to optimize away one runtime branch
|
||||
T val;
|
||||
if(std::is_same<T, rccl_float8>::value)
|
||||
@@ -988,12 +988,12 @@ inline __host__ __device__ T explicit_downcast(Ta a, uint32_t rng)
|
||||
else
|
||||
val.data = rccl_bfloat8::cast_to_bf8_from_f32<stochastic_rounding>(float(a), rng);
|
||||
return val;
|
||||
#else // non gfx940
|
||||
#else // non gfx942
|
||||
return T(float(a),
|
||||
stochastic_rounding ? T::rocblas_hip_f8_rounding_mode::stochastic
|
||||
: T::rocblas_hip_f8_rounding_mode::standard,
|
||||
rng);
|
||||
#endif // __gfx940__
|
||||
#endif // __gfx942__
|
||||
}
|
||||
|
||||
// NOTE NOTE: The above code is good if we don't consider HIP-GEMM code and only consider the quantization
|
||||
|
||||
Reference in New Issue
Block a user