add support for GPUs using wavefront size of 32 (#285)

* add gfx1100 support

Add support for Radeon 7900 GPUs (RX and PRO), and 7800 PRO.

I was contemplating to add gfx1101 and gfx1102 GPUs as well, but those are the lower end models that are more unlikely to be used for compute intensive jobs. In addition, I do not have access to them to test the support.

* update WF_SIZe for different options

Radeon systems use a WarpSize of 32, unlike current Instinct systems,
which use a warp size of 64. For the device side, a gfx specific ifdef
is sufficient. For the host side, we need to query the device
properties.

* adjust functional tests to wf_size of 32

* update unit tests to handle wf_size of 32

* address reviewer comments

[ROCm/rocshmem commit: d0c2845031]
此提交包含在:
Edgar Gabriel
2025-10-22 16:04:58 -05:00
提交者 GitHub
父節點 b771a26916
當前提交 d37af80d7e
共有 19 個檔案被更改,包括 192 行新增56 行删除
+8 -8
查看文件
@@ -46,7 +46,7 @@ __device__ __forceinline__ int uncached_load_ubyte(uint8_t* src) {
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile(
"global_load_ubyte %0 %1 off glc slc \n"
"s_waitcnt vmcnt(0)"
@@ -69,7 +69,7 @@ __device__ __forceinline__ void refresh_volatile_sbyte(volatile int *assigned_va
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile(
"global_load_sbyte %0 %1 off glc slc\n "
"s_waitcnt vmcnt(0)"
@@ -91,7 +91,7 @@ __device__ __forceinline__ void refresh_volatile_dwordx2(volatile uint64_t *assi
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile(
"global_load_dwordx2 %0 %1 off glc slc\n "
"s_waitcnt vmcnt(0)"
@@ -122,7 +122,7 @@ NOWARN(-Wdeprecated-volatile,
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile(
"global_load_dword %0 %1 off glc slc \n"
"s_waitcnt vmcnt(0)"
@@ -142,7 +142,7 @@ NOWARN(-Wdeprecated-volatile,
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile(
"global_load_dwordx2 %0 %1 off glc slc \n"
"s_waitcnt vmcnt(0)"
@@ -191,7 +191,7 @@ __device__ __forceinline__ void store_asm(uint8_t* val, uint8_t* dst,
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile("flat_store_short %0 %1 glc slc" : : "v"(dst), "v"(val16));
#endif
#if defined(__gfx942__) || defined(__gfx950__)
@@ -205,7 +205,7 @@ __device__ __forceinline__ void store_asm(uint8_t* val, uint8_t* dst,
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile("flat_store_dword %0 %1 glc slc" : : "v"(dst), "v"(val32));
#endif
#if defined(__gfx942__) || defined(__gfx950__)
@@ -219,7 +219,7 @@ __device__ __forceinline__ void store_asm(uint8_t* val, uint8_t* dst,
#endif
#if defined(__gfx908__)
#endif
#if defined(__gfx90a__)
#if defined(__gfx90a__) || defined (__gfx1100__)
asm volatile("flat_store_dwordx2 %0 %1 glc slc" : : "v"(dst), "v"(val64));
#endif
#if defined(__gfx942__) || defined(__gfx950__)