Add 'projects/rocr-runtime/' from commit '72061a9024139fa0a99f73f9d3d4deb275670095'

git-subtree-dir: projects/rocr-runtime
git-subtree-mainline: ad0fb25ed5
git-subtree-split: 72061a9024
This commit is contained in:
systems-assistant[bot]
2025-07-22 22:52:49 +00:00
645 changed files with 322673 additions and 0 deletions
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,191 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2017-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
#ifndef _AMDGPU_ASIC_ADDR_H
#define _AMDGPU_ASIC_ADDR_H
#define ATI_VENDOR_ID 0x1002
#define AMD_VENDOR_ID 0x1022
// AMDGPU_VENDOR_IS_AMD(vendorId)
#define AMDGPU_VENDOR_IS_AMD(v) ((v == ATI_VENDOR_ID) || (v == AMD_VENDOR_ID))
#define FAMILY_UNKNOWN 0x00
#define FAMILY_TN 0x69 //# 105 / Trinity APUs
#define FAMILY_SI 0x6E //# 110 / Southern Islands: Tahiti, Pitcairn, CapeVerde, Oland, Hainan
#define FAMILY_CI 0x78 //# 120 / Sea Islands: Bonaire, Hawaii
#define FAMILY_KV 0x7D //# 125 / Kaveri APUs: Spectre, Spooky, Kalindi, Godavari
#define FAMILY_VI 0x82 //# 130 / Volcanic Islands: Iceland, Tonga, Fiji
#define FAMILY_CZ 0x87 //# 135 / Carrizo APUs: Carrizo, Stoney
#define FAMILY_AI 0x8D //# 141 / Vega: 10, 20
#define FAMILY_RV 0x8E //# 142 / Raven
#define FAMILY_NV 0x8F //# 143 / Navi: 10
#define FAMILY_VGH 0x90 //# 144 / Van Gogh
#define FAMILY_NV3 0x91 //# 145 / Navi: 3x
#define FAMILY_GFX1150 0x96
#define FAMILY_GFX1103 0x94
#define FAMILY_RMB 0x92 //# 146 / Rembrandt
#define FAMILY_RPL 0x95 //# 149 / Raphael
#define FAMILY_MDN 0x97 //# 151 / Mendocino
#define FAMILY_GFX12 0x98
// AMDGPU_FAMILY_IS(familyId, familyName)
#define FAMILY_IS(f, fn) (f == FAMILY_##fn)
#define FAMILY_IS_TN(f) FAMILY_IS(f, TN)
#define FAMILY_IS_SI(f) FAMILY_IS(f, SI)
#define FAMILY_IS_CI(f) FAMILY_IS(f, CI)
#define FAMILY_IS_KV(f) FAMILY_IS(f, KV)
#define FAMILY_IS_VI(f) FAMILY_IS(f, VI)
#define FAMILY_IS_POLARIS(f) FAMILY_IS(f, POLARIS)
#define FAMILY_IS_CZ(f) FAMILY_IS(f, CZ)
#define FAMILY_IS_AI(f) FAMILY_IS(f, AI)
#define FAMILY_IS_RV(f) FAMILY_IS(f, RV)
#define FAMILY_IS_NV(f) FAMILY_IS(f, NV)
#define FAMILY_IS_NV3(f) FAMILY_IS(f, NV3)
#define FAMILY_IS_RMB(f) FAMILY_IS(f, RMB)
#define FAMILY_IS_GFX12(f) FAMILY_IS(f, GFX12)
#define AMDGPU_UNKNOWN 0xFF
#define AMDGPU_TAHITI_RANGE 0x05, 0x14 //# 5 <= x < 20
#define AMDGPU_PITCAIRN_RANGE 0x15, 0x28 //# 21 <= x < 40
#define AMDGPU_CAPEVERDE_RANGE 0x29, 0x3C //# 41 <= x < 60
#define AMDGPU_OLAND_RANGE 0x3C, 0x46 //# 60 <= x < 70
#define AMDGPU_HAINAN_RANGE 0x46, 0xFF //# 70 <= x < max
#define AMDGPU_BONAIRE_RANGE 0x14, 0x28 //# 20 <= x < 40
#define AMDGPU_HAWAII_RANGE 0x28, 0x3C //# 40 <= x < 60
#define AMDGPU_SPECTRE_RANGE 0x01, 0x41 //# 1 <= x < 65
#define AMDGPU_SPOOKY_RANGE 0x41, 0x81 //# 65 <= x < 129
#define AMDGPU_KALINDI_RANGE 0x81, 0xA1 //# 129 <= x < 161
#define AMDGPU_GODAVARI_RANGE 0xA1, 0xFF //# 161 <= x < max
#define AMDGPU_ICELAND_RANGE 0x01, 0x14 //# 1 <= x < 20
#define AMDGPU_TONGA_RANGE 0x14, 0x28 //# 20 <= x < 40
#define AMDGPU_FIJI_RANGE 0x3C, 0x50 //# 60 <= x < 80
#define AMDGPU_POLARIS10_RANGE 0x50, 0x5A //# 80 <= x < 90
#define AMDGPU_POLARIS11_RANGE 0x5A, 0x64 //# 90 <= x < 100
#define AMDGPU_POLARIS12_RANGE 0x64, 0x6E //# 100 <= x < 110
#define AMDGPU_VEGAM_RANGE 0x6E, 0xFF //# 110 <= x < max
#define AMDGPU_CARRIZO_RANGE 0x01, 0x21 //# 1 <= x < 33
#define AMDGPU_BRISTOL_RANGE 0x10, 0x21 //# 16 <= x < 33
#define AMDGPU_STONEY_RANGE 0x61, 0xFF //# 97 <= x < max
#define AMDGPU_VEGA10_RANGE 0x01, 0x14 //# 1 <= x < 20
#define AMDGPU_VEGA12_RANGE 0x14, 0x28 //# 20 <= x < 40
#define AMDGPU_VEGA20_RANGE 0x28, 0xFF //# 40 <= x < max
#define AMDGPU_RAVEN_RANGE 0x01, 0x81 //# 1 <= x < 129
#define AMDGPU_RAVEN2_RANGE 0x81, 0x90 //# 129 <= x < 144
#define AMDGPU_RENOIR_RANGE 0x91, 0xFF //# 145 <= x < max
#define AMDGPU_NAVI10_RANGE 0x01, 0x0A //# 1 <= x < 10
#define AMDGPU_NAVI12_RANGE 0x0A, 0x14 //# 10 <= x < 20
#define AMDGPU_NAVI14_RANGE 0x14, 0x28 //# 20 <= x < 40
#define AMDGPU_NAVI21_RANGE 0x28, 0x32 //# 40 <= x < 50
#define AMDGPU_NAVI22_RANGE 0x32, 0x3C //# 50 <= x < 60
#define AMDGPU_NAVI23_RANGE 0x3C, 0x46 //# 60 <= x < 70
#define AMDGPU_NAVI24_RANGE 0x46, 0x50 //# 70 <= x < 80
#define AMDGPU_VANGOGH_RANGE 0x01, 0xFF //# 1 <= x < max
#define AMDGPU_NAVI31_RANGE 0x01, 0x10 //# 01 <= x < 16
#define AMDGPU_NAVI32_RANGE 0x20, 0xFF //# 32 <= x < 255
#define AMDGPU_NAVI33_RANGE 0x10, 0x20 //# 16 <= x < 32
#define AMDGPU_GFX1103_R1_RANGE 0x01, 0x80 //# 1 <= x < 128
#define AMDGPU_GFX1103_R2_RANGE 0x80, 0xC0 //# 128 <= x < 192
#define AMDGPU_GFX1150_RANGE 0x01, 0xFF //# 1 <= x < max
#define AMDGPU_REMBRANDT_RANGE 0x01, 0xFF //# 01 <= x < 255
#define AMDGPU_RAPHAEL_RANGE 0x01, 0xFF //# 1 <= x < max
#define AMDGPU_MENDOCINO_RANGE 0x01, 0xFF //# 1 <= x < max
#define AMDGPU_GFX12_TBD1_RANGE 0x40, 0xFF //# 64 <= x < max
#define AMDGPU_EXPAND_FIX(x) x
#define AMDGPU_RANGE_HELPER(val, min, max) ((val >= min) && (val < max))
#define AMDGPU_IN_RANGE(val, ...) AMDGPU_EXPAND_FIX(AMDGPU_RANGE_HELPER(val, __VA_ARGS__))
// ASICREV_IS(eRevisionId, revisionName)
#define ASICREV_IS(r, rn) AMDGPU_IN_RANGE(r, AMDGPU_##rn##_RANGE)
#define ASICREV_IS_TAHITI_P(r) ASICREV_IS(r, TAHITI)
#define ASICREV_IS_PITCAIRN_PM(r) ASICREV_IS(r, PITCAIRN)
#define ASICREV_IS_CAPEVERDE_M(r) ASICREV_IS(r, CAPEVERDE)
#define ASICREV_IS_OLAND_M(r) ASICREV_IS(r, OLAND)
#define ASICREV_IS_HAINAN_V(r) ASICREV_IS(r, HAINAN)
#define ASICREV_IS_BONAIRE_M(r) ASICREV_IS(r, BONAIRE)
#define ASICREV_IS_HAWAII_P(r) ASICREV_IS(r, HAWAII)
#define ASICREV_IS_SPECTRE(r) ASICREV_IS(r, SPECTRE)
#define ASICREV_IS_SPOOKY(r) ASICREV_IS(r, SPOOKY)
#define ASICREV_IS_KALINDI(r) ASICREV_IS(r, KALINDI)
#define ASICREV_IS_KALINDI_GODAVARI(r) ASICREV_IS(r, GODAVARI)
#define ASICREV_IS_ICELAND_M(r) ASICREV_IS(r, ICELAND)
#define ASICREV_IS_TONGA_P(r) ASICREV_IS(r, TONGA)
#define ASICREV_IS_FIJI_P(r) ASICREV_IS(r, FIJI)
#define ASICREV_IS_POLARIS10_P(r) ASICREV_IS(r, POLARIS10)
#define ASICREV_IS_POLARIS11_M(r) ASICREV_IS(r, POLARIS11)
#define ASICREV_IS_POLARIS12_V(r) ASICREV_IS(r, POLARIS12)
#define ASICREV_IS_VEGAM_P(r) ASICREV_IS(r, VEGAM)
#define ASICREV_IS_CARRIZO(r) ASICREV_IS(r, CARRIZO)
#define ASICREV_IS_CARRIZO_BRISTOL(r) ASICREV_IS(r, BRISTOL)
#define ASICREV_IS_STONEY(r) ASICREV_IS(r, STONEY)
#define ASICREV_IS_VEGA10_M(r) ASICREV_IS(r, VEGA10)
#define ASICREV_IS_VEGA10_P(r) ASICREV_IS(r, VEGA10)
#define ASICREV_IS_VEGA12_P(r) ASICREV_IS(r, VEGA12)
#define ASICREV_IS_VEGA12_p(r) ASICREV_IS(r, VEGA12)
#define ASICREV_IS_VEGA20_P(r) ASICREV_IS(r, VEGA20)
#define ASICREV_IS_RAVEN(r) ASICREV_IS(r, RAVEN)
#define ASICREV_IS_RAVEN2(r) ASICREV_IS(r, RAVEN2)
#define ASICREV_IS_RENOIR(r) ASICREV_IS(r, RENOIR)
#define ASICREV_IS_NAVI10_P(r) ASICREV_IS(r, NAVI10)
#define ASICREV_IS_NAVI12_P(r) ASICREV_IS(r, NAVI12)
#define ASICREV_IS_NAVI14_M(r) ASICREV_IS(r, NAVI14)
#define ASICREV_IS_NAVI21_M(r) ASICREV_IS(r, NAVI21)
#define ASICREV_IS_NAVI22_P(r) ASICREV_IS(r, NAVI22)
#define ASICREV_IS_NAVI23_P(r) ASICREV_IS(r, NAVI23)
#define ASICREV_IS_NAVI24_P(r) ASICREV_IS(r, NAVI24)
#define ASICREV_IS_VANGOGH(r) ASICREV_IS(r, VANGOGH)
#define ASICREV_IS_NAVI31_P(r) ASICREV_IS(r, NAVI31)
#define ASICREV_IS_NAVI32_P(r) ASICREV_IS(r, NAVI32)
#define ASICREV_IS_NAVI33_P(r) ASICREV_IS(r, NAVI33)
#define ASICREV_IS_GFX1103_R1(r) ASICREV_IS(r, GFX1103_R1)
#define ASICREV_IS_GFX1103_R2(r) ASICREV_IS(r, GFX1103_R2)
#define ASICREV_IS_GFX1150(r) ASICREV_IS(r, GFX1150)
#define ASICREV_IS_REMBRANDT(r) ASICREV_IS(r, REMBRANDT)
#define ASICREV_IS_RAPHAEL(r) ASICREV_IS(r, RAPHAEL)
#define ASICREV_IS_MENDOCINO(r) ASICREV_IS(r, MENDOCINO)
#define ASICREV_IS_GFX12_TBD1_P(r) ASICREV_IS(r, GFX12_TBD1)
#endif // _AMDGPU_ASIC_ADDR_H
@@ -0,0 +1,52 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
#if !defined (__GFX10_GB_REG_H__)
#define __GFX10_GB_REG_H__
/*
* gfx10_gb_reg.h
*
* Register Spec Release: 1.0
*
*/
//
// Make sure the necessary endian defines are there.
//
#if defined(LITTLEENDIAN_CPU)
#elif defined(BIGENDIAN_CPU)
#else
#error "BIGENDIAN_CPU or LITTLEENDIAN_CPU must be defined"
#endif
union GB_ADDR_CONFIG_GFX10
{
struct
{
#if defined(LITTLEENDIAN_CPU)
unsigned int NUM_PIPES : 3;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int NUM_PKRS : 3;
unsigned int : 21;
#elif defined(BIGENDIAN_CPU)
unsigned int : 21;
unsigned int NUM_PKRS : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int NUM_PIPES : 3;
#endif
} bitfields, bits;
unsigned int u32All;
int i32All;
float f32All;
};
#endif
@@ -0,0 +1,60 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
#if !defined (__GFX11_GB_REG_H__)
#define __GFX11_GB_REG_H__
/*
* gfx11_gb_reg.h
*
* Register Spec Release: 1.0
*
*/
//
// Make sure the necessary endian defines are there.
//
#if defined(LITTLEENDIAN_CPU)
#elif defined(BIGENDIAN_CPU)
#else
#error "BIGENDIAN_CPU or LITTLEENDIAN_CPU must be defined"
#endif
union GB_ADDR_CONFIG_GFX11
{
struct
{
#if defined(LITTLEENDIAN_CPU)
unsigned int NUM_PIPES : 3;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int NUM_PKRS : 3;
unsigned int : 8;
unsigned int NUM_SHADER_ENGINES : 2;
unsigned int : 5;
unsigned int NUM_RB_PER_SE : 2;
unsigned int : 4;
#elif defined(BIGENDIAN_CPU)
unsigned int : 4;
unsigned int NUM_RB_PER_SE : 2;
unsigned int : 5;
unsigned int NUM_SHADER_ENGINES : 2;
unsigned int : 8;
unsigned int NUM_PKRS : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int NUM_PIPES : 3;
#endif
} bitfields, bits;
unsigned int u32All;
int i32All;
float f32All;
};
#endif
@@ -0,0 +1,57 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2023 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
#if !defined (__GFX12_GB_REG_H__)
#define __GFX12_GB_REG_H__
/*
* gfx12_gb_reg.h
*
* Register Spec Release: 1.0
*
*/
//
// Make sure the necessary endian defines are there.
//
#if defined(LITTLEENDIAN_CPU)
#elif defined(BIGENDIAN_CPU)
#else
#error "BIGENDIAN_CPU or LITTLEENDIAN_CPU must be defined"
#endif
union GB_ADDR_CONFIG_GFX12 {
struct {
#if defined(LITTLEENDIAN_CPU)
unsigned int NUM_PIPES : 3;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int NUM_PKRS : 3;
unsigned int : 8;
unsigned int NUM_SHADER_ENGINES : 4;
unsigned int : 3;
unsigned int NUM_RB_PER_SE : 2;
unsigned int : 4;
#elif defined(BIGENDIAN_CPU)
unsigned int : 4;
unsigned int NUM_RB_PER_SE : 2;
unsigned int : 3;
unsigned int NUM_SHADER_ENGINES : 4;
unsigned int : 8;
unsigned int NUM_PKRS : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int NUM_PIPES : 3;
#endif
} bitfields, bits;
unsigned int u32All;
int i32All;
float f32All;
};
#endif
@@ -0,0 +1,70 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
#if !defined (__GFX9_GB_REG_H__)
#define __GFX9_GB_REG_H__
/*
* gfx9_gb_reg.h
*
* Register Spec Release: 1.0
*
*/
//
// Make sure the necessary endian defines are there.
//
#if defined(LITTLEENDIAN_CPU)
#elif defined(BIGENDIAN_CPU)
#else
#error "BIGENDIAN_CPU or LITTLEENDIAN_CPU must be defined"
#endif
union GB_ADDR_CONFIG_GFX9 {
struct {
#if defined(LITTLEENDIAN_CPU)
unsigned int NUM_PIPES : 3;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int BANK_INTERLEAVE_SIZE : 3;
unsigned int : 1;
unsigned int NUM_BANKS : 3;
unsigned int : 1;
unsigned int SHADER_ENGINE_TILE_SIZE : 3;
unsigned int NUM_SHADER_ENGINES : 2;
unsigned int NUM_GPUS : 3;
unsigned int MULTI_GPU_TILE_SIZE : 2;
unsigned int NUM_RB_PER_SE : 2;
unsigned int ROW_SIZE : 2;
unsigned int NUM_LOWER_PIPES : 1;
unsigned int SE_ENABLE : 1;
#elif defined(BIGENDIAN_CPU)
unsigned int SE_ENABLE : 1;
unsigned int NUM_LOWER_PIPES : 1;
unsigned int ROW_SIZE : 2;
unsigned int NUM_RB_PER_SE : 2;
unsigned int MULTI_GPU_TILE_SIZE : 2;
unsigned int NUM_GPUS : 3;
unsigned int NUM_SHADER_ENGINES : 2;
unsigned int SHADER_ENGINE_TILE_SIZE : 3;
unsigned int : 1;
unsigned int NUM_BANKS : 3;
unsigned int : 1;
unsigned int BANK_INTERLEAVE_SIZE : 3;
unsigned int MAX_COMPRESSED_FRAGS : 2;
unsigned int PIPE_INTERLEAVE_SIZE : 3;
unsigned int NUM_PIPES : 3;
#endif
} bitfields, bits;
unsigned int u32All;
signed int i32All;
float f32All;
};
#endif
@@ -0,0 +1,194 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
#if !defined (__SI_GB_REG_H__)
#define __SI_GB_REG_H__
/*****************************************************************************************************************
*
* si_gb_reg.h
*
* Register Spec Release: Chip Spec 0.28
*
*****************************************************************************************************************/
//
// Make sure the necessary endian defines are there.
//
#if defined(LITTLEENDIAN_CPU)
#elif defined(BIGENDIAN_CPU)
#else
#error "BIGENDIAN_CPU or LITTLEENDIAN_CPU must be defined"
#endif
/*
* GB_ADDR_CONFIG struct
*/
#if defined(LITTLEENDIAN_CPU)
typedef struct _GB_ADDR_CONFIG_T {
unsigned int num_pipes : 3;
unsigned int : 1;
unsigned int pipe_interleave_size : 3;
unsigned int : 1;
unsigned int bank_interleave_size : 3;
unsigned int : 1;
unsigned int num_shader_engines : 2;
unsigned int : 2;
unsigned int shader_engine_tile_size : 3;
unsigned int : 1;
unsigned int num_gpus : 3;
unsigned int : 1;
unsigned int multi_gpu_tile_size : 2;
unsigned int : 2;
unsigned int row_size : 2;
unsigned int num_lower_pipes : 1;
unsigned int : 1;
} GB_ADDR_CONFIG_T;
#elif defined(BIGENDIAN_CPU)
typedef struct _GB_ADDR_CONFIG_T {
unsigned int : 1;
unsigned int num_lower_pipes : 1;
unsigned int row_size : 2;
unsigned int : 2;
unsigned int multi_gpu_tile_size : 2;
unsigned int : 1;
unsigned int num_gpus : 3;
unsigned int : 1;
unsigned int shader_engine_tile_size : 3;
unsigned int : 2;
unsigned int num_shader_engines : 2;
unsigned int : 1;
unsigned int bank_interleave_size : 3;
unsigned int : 1;
unsigned int pipe_interleave_size : 3;
unsigned int : 1;
unsigned int num_pipes : 3;
} GB_ADDR_CONFIG_T;
#endif
#if defined(LITTLEENDIAN_CPU)
typedef struct _GB_ADDR_CONFIG_N {
unsigned int num_pipes : 3;
unsigned int pipe_interleave_size : 3;
unsigned int max_compressed_frags : 2;
unsigned int bank_interleave_size : 3;
unsigned int : 1;
unsigned int num_banks : 3;
unsigned int : 1;
unsigned int shader_engine_tile_size : 3;
unsigned int num_shader_engines : 2;
unsigned int num_gpus : 3;
unsigned int multi_gpu_tile_size : 2;
unsigned int num_rb_per_se : 2;
unsigned int row_size : 2;
unsigned int num_lower_pipes : 1;
unsigned int se_enable : 1;
} GB_ADDR_CONFIG_N;
#elif defined(BIGENDIAN_CPU)
typedef struct _GB_ADDR_CONFIG_N {
unsigned int se_enable : 1;
unsigned int num_lower_pipes : 1;
unsigned int row_size : 2;
unsigned int num_rb_per_se : 2;
unsigned int multi_gpu_tile_size : 2;
unsigned int num_gpus : 3;
unsigned int num_shader_engines : 2;
unsigned int shader_engine_tile_size : 3;
unsigned int : 1;
unsigned int num_banks : 3;
unsigned int : 1;
unsigned int bank_interleave_size : 3;
unsigned int max_compressed_frags : 2;
unsigned int pipe_interleave_size : 3;
unsigned int num_pipes : 3;
} GB_ADDR_CONFIG_N;
#endif
typedef union {
unsigned int val : 32;
GB_ADDR_CONFIG_T f;
GB_ADDR_CONFIG_N n;
} GB_ADDR_CONFIG;
#if defined(LITTLEENDIAN_CPU)
typedef struct _GB_TILE_MODE_T {
unsigned int micro_tile_mode : 2;
unsigned int array_mode : 4;
unsigned int pipe_config : 5;
unsigned int tile_split : 3;
unsigned int bank_width : 2;
unsigned int bank_height : 2;
unsigned int macro_tile_aspect : 2;
unsigned int num_banks : 2;
unsigned int micro_tile_mode_new : 3;
unsigned int sample_split : 2;
unsigned int alt_pipe_config : 5;
} GB_TILE_MODE_T;
typedef struct _GB_MACROTILE_MODE_T {
unsigned int bank_width : 2;
unsigned int bank_height : 2;
unsigned int macro_tile_aspect : 2;
unsigned int num_banks : 2;
unsigned int alt_bank_height : 2;
unsigned int alt_macro_tile_aspect : 2;
unsigned int alt_num_banks : 2;
unsigned int : 18;
} GB_MACROTILE_MODE_T;
#elif defined(BIGENDIAN_CPU)
typedef struct _GB_TILE_MODE_T {
unsigned int alt_pipe_config : 5;
unsigned int sample_split : 2;
unsigned int micro_tile_mode_new : 3;
unsigned int num_banks : 2;
unsigned int macro_tile_aspect : 2;
unsigned int bank_height : 2;
unsigned int bank_width : 2;
unsigned int tile_split : 3;
unsigned int pipe_config : 5;
unsigned int array_mode : 4;
unsigned int micro_tile_mode : 2;
} GB_TILE_MODE_T;
typedef struct _GB_MACROTILE_MODE_T {
unsigned int : 18;
unsigned int alt_num_banks : 2;
unsigned int alt_macro_tile_aspect : 2;
unsigned int alt_bank_height : 2;
unsigned int num_banks : 2;
unsigned int macro_tile_aspect : 2;
unsigned int bank_height : 2;
unsigned int bank_width : 2;
} GB_MACROTILE_MODE_T;
#endif
typedef union {
unsigned int val : 32;
GB_TILE_MODE_T f;
} GB_TILE_MODE;
typedef union {
unsigned int val : 32;
GB_MACROTILE_MODE_T f;
} GB_MACROTILE_MODE;
#endif
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,263 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
****************************************************************************************************
* @file addrelemlib.h
* @brief Contains the class for element/pixel related functions.
****************************************************************************************************
*/
#ifndef __ELEM_LIB_H__
#define __ELEM_LIB_H__
#include "addrinterface.h"
#include "addrobject.h"
#include "addrcommon.h"
namespace rocr {
namespace Addr
{
class Lib;
// The masks for property bits within the Properties INT_32
union ComponentFlags
{
struct
{
UINT_32 byteAligned : 1; ///< all components are byte aligned
UINT_32 exportNorm : 1; ///< components support R6xx NORM compression
UINT_32 floatComp : 1; ///< there is at least one floating point component
};
UINT_32 value;
};
// Copy from legacy lib's NumberType
enum NumberType
{
// The following number types have the range [-1..1]
ADDR_NO_NUMBER, // This component doesn't exist and has no default value
ADDR_EPSILON, // Force component value to integer 0x00000001
ADDR_ZERO, // Force component value to integer 0x00000000
ADDR_ONE, // Force component value to floating point 1.0
// Above values don't have any bits per component (keep ADDR_ONE the last of these)
ADDR_UNORM, // Unsigned normalized (repeating fraction) full precision
ADDR_SNORM, // Signed normalized (repeating fraction) full precision
ADDR_GAMMA, // Gamma-corrected, full precision
ADDR_UNORM_R5XXRB, // Unsigned normalized (repeating fraction) for r5xx RB
ADDR_SNORM_R5XXRB, // Signed normalized (repeating fraction) for r5xx RB
ADDR_GAMMA_R5XXRB, // Gamma-corrected for r5xx RB (note: unnormalized value)
ADDR_UNORM_R5XXBC, // Unsigned normalized (repeating fraction) for r5xx BC
ADDR_SNORM_R5XXBC, // Signed normalized (repeating fraction) for r5xx BC
ADDR_GAMMA_R5XXBC, // Gamma-corrected for r5xx BC (note: unnormalized value)
ADDR_UNORM_R6XX, // Unsigned normalized (repeating fraction) for R6xx
ADDR_UNORM_R6XXDB, // Unorms for 24-bit depth: one value differs from ADDR_UNORM_R6XX
ADDR_SNORM_R6XX, // Signed normalized (repeating fraction) for R6xx
ADDR_GAMMA8_R6XX, // Gamma-corrected for r6xx
ADDR_GAMMA8_R7XX_TP, // Gamma-corrected for r7xx TP 12bit unorm 8.4.
ADDR_U4FLOATC, // Unsigned float: 4-bit exponent, bias=15, no NaN, clamp [0..1]
ADDR_GAMMA_4SEG, // Gamma-corrected, four segment approximation
ADDR_U0FIXED, // Unsigned 0.N-bit fixed point
// The following number types have large ranges (LEAVE ADDR_USCALED first or fix Finish routine)
ADDR_USCALED, // Unsigned integer converted to/from floating point
ADDR_SSCALED, // Signed integer converted to/from floating point
ADDR_USCALED_R5XXRB, // Unsigned integer to/from floating point for r5xx RB
ADDR_SSCALED_R5XXRB, // Signed integer to/from floating point for r5xx RB
ADDR_UINT_BITS, // Keep in unsigned integer form, clamped to specified range
ADDR_SINT_BITS, // Keep in signed integer form, clamped to specified range
ADDR_UINTBITS, // @@ remove Keep in unsigned integer form, use modulus to reduce bits
ADDR_SINTBITS, // @@ remove Keep in signed integer form, use modulus to reduce bits
// The following number types and ADDR_U4FLOATC have exponents
// (LEAVE ADDR_S8FLOAT first or fix Finish routine)
ADDR_S8FLOAT, // Signed floating point with 8-bit exponent, bias=127
ADDR_S8FLOAT32, // 32-bit IEEE float, passes through NaN values
ADDR_S5FLOAT, // Signed floating point with 5-bit exponent, bias=15
ADDR_S5FLOATM, // Signed floating point with 5-bit exponent, bias=15, no NaN/Inf
ADDR_U5FLOAT, // Signed floating point with 5-bit exponent, bias=15
ADDR_U3FLOATM, // Unsigned floating point with 3-bit exponent, bias=3
ADDR_S5FIXED, // Signed 5.N-bit fixed point, with rounding
ADDR_END_NUMBER // Used for range comparisons
};
// Copy from legacy lib's AddrElement
enum ElemMode
{
// These formats allow both packing an unpacking
ADDR_ROUND_BY_HALF, // add 1/2 and truncate when packing this element
ADDR_ROUND_TRUNCATE, // truncate toward 0 for sign/mag, else toward neg
ADDR_ROUND_DITHER, // Pack by dithering -- requires (x,y) position
// These formats only allow unpacking, no packing
ADDR_UNCOMPRESSED, // Elements are not compressed: one data element per pixel/texel
ADDR_EXPANDED, // Elements are split up and stored in multiple data elements
ADDR_PACKED_STD, // Elements are compressed into ExpandX by ExpandY data elements
ADDR_PACKED_REV, // Like ADDR_PACKED, but X order of pixels is reverved
ADDR_PACKED_GBGR, // Elements are compressed 4:2:2 in G1B_G0R order (high to low)
ADDR_PACKED_BGRG, // Elements are compressed 4:2:2 in BG1_RG0 order (high to low)
ADDR_PACKED_BC1, // Each data element is uncompressed to a 4x4 pixel/texel array
ADDR_PACKED_BC2, // Each data element is uncompressed to a 4x4 pixel/texel array
ADDR_PACKED_BC3, // Each data element is uncompressed to a 4x4 pixel/texel array
ADDR_PACKED_BC4, // Each data element is uncompressed to a 4x4 pixel/texel array
ADDR_PACKED_BC5, // Each data element is uncompressed to a 4x4 pixel/texel array
ADDR_PACKED_ETC2_64BPP, // ETC2 formats that use 64bpp to represent each 4x4 block
ADDR_PACKED_ETC2_128BPP, // ETC2 formats that use 128bpp to represent each 4x4 block
ADDR_PACKED_ASTC, // Various ASTC formats, all are 128bpp with varying block sizes
// These formats provide various kinds of compression
ADDR_ZPLANE_R5XX, // Compressed Zplane using r5xx architecture format
ADDR_ZPLANE_R6XX, // Compressed Zplane using r6xx architecture format
//@@ Fill in the compression modes
ADDR_END_ELEMENT // Used for range comparisons
};
enum DepthPlanarType
{
ADDR_DEPTH_PLANAR_NONE = 0, // No plane z/stencl
ADDR_DEPTH_PLANAR_R600 = 1, // R600 z and stencil planes are store within a tile
ADDR_DEPTH_PLANAR_R800 = 2, // R800 has separate z and stencil planes
};
/**
****************************************************************************************************
* PixelFormatInfo
*
* @brief
* Per component info
*
****************************************************************************************************
*/
struct PixelFormatInfo
{
UINT_32 compBit[4];
NumberType numType[4];
UINT_32 compStart[4];
ElemMode elemMode;
UINT_32 comps; ///< Number of components
};
/**
****************************************************************************************************
* @brief This class contains asic indepentent element related attributes and operations
****************************************************************************************************
*/
class ElemLib : public Object
{
protected:
ElemLib(Lib* pAddrLib);
public:
/// Makes this class virtual
virtual ~ElemLib();
static ElemLib* Create(
const Lib* pAddrLib);
/// The implementation is only for R6xx/R7xx, so make it virtual in case we need for R8xx
BOOL_32 PixGetExportNorm(
AddrColorFormat colorFmt,
AddrSurfaceNumber numberFmt, AddrSurfaceSwap swap) const;
/// Below method are asic independent, so make them just static.
/// Remove static if we need different operation in hwl.
VOID Flt32ToDepthPixel(
AddrDepthFormat format, const ADDR_FLT_32 comps[2], UINT_8 *pPixel) const;
VOID Flt32ToColorPixel(
AddrColorFormat format, AddrSurfaceNumber surfNum, AddrSurfaceSwap surfSwap,
const ADDR_FLT_32 comps[4], UINT_8 *pPixel) const;
static VOID Flt32sToInt32s(
ADDR_FLT_32 value, UINT_32 bits, NumberType numberType, UINT_32* pResult);
static VOID Int32sToPixel(
UINT_32 numComps, UINT_32* pComps, UINT_32* pCompBits, UINT_32* pCompStart,
ComponentFlags properties, UINT_32 resultBits, UINT_8* pPixel);
VOID PixGetColorCompInfo(
AddrColorFormat format, AddrSurfaceNumber number, AddrSurfaceSwap swap,
PixelFormatInfo* pInfo) const;
VOID PixGetDepthCompInfo(
AddrDepthFormat format, PixelFormatInfo* pInfo) const;
UINT_32 GetBitsPerPixel(
AddrFormat format, ElemMode* pElemMode = NULL,
UINT_32* pExpandX = NULL, UINT_32* pExpandY = NULL, UINT_32* pBitsUnused = NULL);
static VOID SetClearComps(
ADDR_FLT_32 comps[4], BOOL_32 clearColor, BOOL_32 float32);
VOID AdjustSurfaceInfo(
ElemMode elemMode, UINT_32 expandX, UINT_32 expandY,
UINT_32* pBpp, UINT_32* pBasePitch, UINT_32* pWidth, UINT_32* pHeight);
VOID RestoreSurfaceInfo(
ElemMode elemMode, UINT_32 expandX, UINT_32 expandY,
UINT_32* pBpp, UINT_32* pWidth, UINT_32* pHeight);
/// Checks if depth and stencil are planar inside a tile
BOOL_32 IsDepthStencilTilePlanar()
{
return (m_depthPlanarType == ADDR_DEPTH_PLANAR_R600) ? TRUE : FALSE;
}
/// Sets m_configFlags, copied from AddrLib
VOID SetConfigFlags(ConfigFlags flags)
{
m_configFlags = flags;
}
static BOOL_32 IsCompressed(AddrFormat format);
static BOOL_32 IsBlockCompressed(AddrFormat format);
static BOOL_32 IsExpand3x(AddrFormat format);
static BOOL_32 IsMacroPixelPacked(AddrFormat format);
protected:
static VOID GetCompBits(
UINT_32 c0, UINT_32 c1, UINT_32 c2, UINT_32 c3,
PixelFormatInfo* pInfo,
ElemMode elemMode = ADDR_ROUND_BY_HALF);
static VOID GetCompType(
AddrColorFormat format, AddrSurfaceNumber numType,
PixelFormatInfo* pInfo);
static VOID GetCompSwap(
AddrSurfaceSwap swap, PixelFormatInfo* pInfo);
static VOID SwapComps(
UINT_32 c0, UINT_32 c1, PixelFormatInfo* pInfo);
private:
UINT_32 m_fp16ExportNorm; ///< If allow FP16 to be reported as EXPORT_NORM
DepthPlanarType m_depthPlanarType;
ConfigFlags m_configFlags; ///< Copy of AddrLib's configFlags
Addr::Lib* const m_pAddrLib; ///< Pointer to parent addrlib instance
};
} //Addr
} //namespace rocr
#endif
@@ -0,0 +1,644 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
****************************************************************************************************
* @file addrlib.cpp
* @brief Contains the implementation for the Addr::Lib class.
****************************************************************************************************
*/
#include "addrinterface.h"
#include "addrlib.h"
#include "addrcommon.h"
#if defined(__APPLE__)
UINT_32 div64_32(UINT_64 n, UINT_32 base)
{
UINT_64 rem = n;
UINT_64 b = base;
UINT_64 res, d = 1;
UINT_32 high = rem >> 32;
res = 0;
if (high >= base)
{
high /= base;
res = (UINT_64) high << 32;
rem -= (UINT_64) (high * base) << 32;
}
while (((INT_64)b > 0) && (b < rem))
{
b = b + b;
d = d + d;
}
do
{
if (rem >= b)
{
rem -= b;
res += d;
}
b >>= 1;
d >>= 1;
} while (d);
n = res;
return rem;
}
extern "C"
UINT_32 __umoddi3(UINT_64 n, UINT_32 base)
{
return div64_32(n, base);
}
#endif // __APPLE__
namespace rocr {
namespace Addr
{
////////////////////////////////////////////////////////////////////////////////////////////////////
// Constructor/Destructor
////////////////////////////////////////////////////////////////////////////////////////////////////
/**
****************************************************************************************************
* Lib::Lib
*
* @brief
* Constructor for the AddrLib class
*
****************************************************************************************************
*/
Lib::Lib() :
m_chipFamily(ADDR_CHIP_FAMILY_IVLD),
m_chipRevision(0),
m_version(ADDRLIB_VERSION),
m_pipes(0),
m_banks(0),
m_pipeInterleaveBytes(0),
m_rowSize(0),
m_minPitchAlignPixels(1),
m_maxSamples(8),
m_maxBaseAlign(0),
m_maxMetaBaseAlign(0),
m_pElemLib(NULL)
{
m_configFlags.value = 0;
}
/**
****************************************************************************************************
* Lib::Lib
*
* @brief
* Constructor for the AddrLib class with hClient as parameter
*
****************************************************************************************************
*/
Lib::Lib(const Client* pClient) :
Object(pClient),
m_chipFamily(ADDR_CHIP_FAMILY_IVLD),
m_chipRevision(0),
m_version(ADDRLIB_VERSION),
m_pipes(0),
m_banks(0),
m_pipeInterleaveBytes(0),
m_rowSize(0),
m_minPitchAlignPixels(1),
m_maxSamples(8),
m_maxBaseAlign(0),
m_maxMetaBaseAlign(0),
m_pElemLib(NULL)
{
m_configFlags.value = 0;
}
/**
****************************************************************************************************
* Lib::~AddrLib
*
* @brief
* Destructor for the AddrLib class
*
****************************************************************************************************
*/
Lib::~Lib()
{
if (m_pElemLib)
{
delete m_pElemLib;
m_pElemLib = NULL;
}
}
////////////////////////////////////////////////////////////////////////////////////////////////////
// Initialization/Helper
////////////////////////////////////////////////////////////////////////////////////////////////////
/**
****************************************************************************************************
* Lib::Create
*
* @brief
* Creates and initializes AddrLib object.
*
* @return
* ADDR_E_RETURNCODE
****************************************************************************************************
*/
ADDR_E_RETURNCODE Lib::Create(
const ADDR_CREATE_INPUT* pCreateIn, ///< [in] pointer to ADDR_CREATE_INPUT
ADDR_CREATE_OUTPUT* pCreateOut) ///< [out] pointer to ADDR_CREATE_OUTPUT
{
Lib* pLib = NULL;
ADDR_E_RETURNCODE returnCode = ADDR_OK;
if (pCreateIn->createFlags.fillSizeFields == TRUE)
{
if ((pCreateIn->size != sizeof(ADDR_CREATE_INPUT)) ||
(pCreateOut->size != sizeof(ADDR_CREATE_OUTPUT)))
{
returnCode = ADDR_PARAMSIZEMISMATCH;
}
}
if ((returnCode == ADDR_OK) &&
(pCreateIn->callbacks.allocSysMem != NULL) &&
(pCreateIn->callbacks.freeSysMem != NULL))
{
Client client = {
pCreateIn->hClient,
pCreateIn->callbacks
};
switch (pCreateIn->chipEngine)
{
case CIASICIDGFXENGINE_ARCTICISLAND:
switch (pCreateIn->chipFamily)
{
case FAMILY_AI:
case FAMILY_RV:
pLib = Gfx9HwlInit(&client);
break;
case FAMILY_NV:
case FAMILY_VGH:
case FAMILY_RMB:
case FAMILY_RPL:
case FAMILY_MDN:
pLib = Gfx10HwlInit(&client);
break;
case FAMILY_NV3:
case FAMILY_GFX1150:
case FAMILY_GFX1103:
pLib = Gfx11HwlInit(&client);
break;
case FAMILY_GFX12:
pLib = Gfx12HwlInit(&client);
break;
default:
ADDR_ASSERT_ALWAYS();
break;
}
break;
default:
ADDR_ASSERT_ALWAYS();
break;
}
}
if(pLib == NULL)
{
returnCode = ADDR_OUTOFMEMORY;
}
if (pLib != NULL)
{
BOOL_32 initValid;
// Pass createFlags to configFlags first since these flags may be overwritten
pLib->m_configFlags.noCubeMipSlicesPad = pCreateIn->createFlags.noCubeMipSlicesPad;
pLib->m_configFlags.fillSizeFields = pCreateIn->createFlags.fillSizeFields;
pLib->m_configFlags.useTileIndex = pCreateIn->createFlags.useTileIndex;
pLib->m_configFlags.useCombinedSwizzle = pCreateIn->createFlags.useCombinedSwizzle;
pLib->m_configFlags.checkLast2DLevel = pCreateIn->createFlags.checkLast2DLevel;
pLib->m_configFlags.useHtileSliceAlign = pCreateIn->createFlags.useHtileSliceAlign;
pLib->m_configFlags.allowLargeThickTile = pCreateIn->createFlags.allowLargeThickTile;
pLib->m_configFlags.forceDccAndTcCompat = pCreateIn->createFlags.forceDccAndTcCompat;
pLib->m_configFlags.nonPower2MemConfig = pCreateIn->createFlags.nonPower2MemConfig;
pLib->m_configFlags.enableAltTiling = pCreateIn->createFlags.enableAltTiling;
pLib->m_configFlags.disableLinearOpt = FALSE;
pLib->SetChipFamily(pCreateIn->chipFamily, pCreateIn->chipRevision);
pLib->SetMinPitchAlignPixels(pCreateIn->minPitchAlignPixels);
// Global parameters initialized and remaining configFlags bits are set as well
initValid = pLib->HwlInitGlobalParams(pCreateIn);
if (initValid)
{
pLib->m_pElemLib = ElemLib::Create(pLib);
}
else
{
pLib->m_pElemLib = NULL; // Don't go on allocating element lib
returnCode = ADDR_INVALIDGBREGVALUES;
}
if (pLib->m_pElemLib == NULL)
{
delete pLib;
pLib = NULL;
returnCode = ADDR_OUTOFMEMORY;
ADDR_ASSERT_ALWAYS();
}
else
{
pLib->m_pElemLib->SetConfigFlags(pLib->m_configFlags);
}
}
pCreateOut->hLib = pLib;
if ((pLib != NULL) &&
(returnCode == ADDR_OK))
{
pCreateOut->numEquations =
pLib->HwlGetEquationTableInfo(&pCreateOut->pEquationTable);
pLib->SetMaxAlignments();
}
return returnCode;
}
/**
****************************************************************************************************
* Lib::SetChipFamily
*
* @brief
* Convert familyID defined in atiid.h to ChipFamily and set m_chipFamily/m_chipRevision
* @return
* N/A
****************************************************************************************************
*/
VOID Lib::SetChipFamily(
UINT_32 uChipFamily, ///< [in] chip family defined in atiih.h
UINT_32 uChipRevision) ///< [in] chip revision defined in "asic_family"_id.h
{
ChipFamily family = HwlConvertChipFamily(uChipFamily, uChipRevision);
ADDR_ASSERT(family != ADDR_CHIP_FAMILY_IVLD);
m_chipFamily = family;
m_chipRevision = uChipRevision;
}
/**
****************************************************************************************************
* Lib::SetMinPitchAlignPixels
*
* @brief
* Set m_minPitchAlignPixels with input param
*
* @return
* N/A
****************************************************************************************************
*/
VOID Lib::SetMinPitchAlignPixels(
UINT_32 minPitchAlignPixels) ///< [in] minmum pitch alignment in pixels
{
m_minPitchAlignPixels = (minPitchAlignPixels == 0) ? 1 : minPitchAlignPixels;
}
/**
****************************************************************************************************
* Lib::SetMaxAlignments
*
* @brief
* Set max alignments
*
* @return
* N/A
****************************************************************************************************
*/
VOID Lib::SetMaxAlignments()
{
m_maxBaseAlign = HwlComputeMaxBaseAlignments();
m_maxMetaBaseAlign = HwlComputeMaxMetaBaseAlignments();
}
/**
****************************************************************************************************
* Lib::GetLib
*
* @brief
* Get AddrLib pointer
*
* @return
* An AddrLib class pointer
****************************************************************************************************
*/
Lib* Lib::GetLib(
ADDR_HANDLE hLib) ///< [in] handle of ADDR_HANDLE
{
return static_cast<Addr::Lib*>(hLib);
}
/**
****************************************************************************************************
* Lib::GetMaxAlignments
*
* @brief
* Gets maximum alignments for data surface (include FMask)
*
* @return
* ADDR_E_RETURNCODE
****************************************************************************************************
*/
ADDR_E_RETURNCODE Lib::GetMaxAlignments(
ADDR_GET_MAX_ALIGNMENTS_OUTPUT* pOut ///< [out] output structure
) const
{
ADDR_E_RETURNCODE returnCode = ADDR_OK;
if (GetFillSizeFieldsFlags() == TRUE)
{
if (pOut->size != sizeof(ADDR_GET_MAX_ALIGNMENTS_OUTPUT))
{
returnCode = ADDR_PARAMSIZEMISMATCH;
}
}
if (returnCode == ADDR_OK)
{
if (m_maxBaseAlign != 0)
{
pOut->baseAlign = m_maxBaseAlign;
}
else
{
returnCode = ADDR_NOTIMPLEMENTED;
}
}
return returnCode;
}
/**
****************************************************************************************************
* Lib::GetMaxMetaAlignments
*
* @brief
* Gets maximum alignments for metadata (CMask, DCC and HTile)
*
* @return
* ADDR_E_RETURNCODE
****************************************************************************************************
*/
ADDR_E_RETURNCODE Lib::GetMaxMetaAlignments(
ADDR_GET_MAX_ALIGNMENTS_OUTPUT* pOut ///< [out] output structure
) const
{
ADDR_E_RETURNCODE returnCode = ADDR_OK;
if (GetFillSizeFieldsFlags() == TRUE)
{
if (pOut->size != sizeof(ADDR_GET_MAX_ALIGNMENTS_OUTPUT))
{
returnCode = ADDR_PARAMSIZEMISMATCH;
}
}
if (returnCode == ADDR_OK)
{
if (m_maxMetaBaseAlign != 0)
{
pOut->baseAlign = m_maxMetaBaseAlign;
}
else
{
returnCode = ADDR_NOTIMPLEMENTED;
}
}
return returnCode;
}
/**
****************************************************************************************************
* Lib::Bits2Number
*
* @brief
* Cat a array of binary bit to a number
*
* @return
* The number combined with the array of bits
****************************************************************************************************
*/
UINT_32 Lib::Bits2Number(
UINT_32 bitNum, ///< [in] how many bits
...) ///< [in] varaible bits value starting from MSB
{
UINT_32 number = 0;
UINT_32 i;
va_list bits_ptr;
va_start(bits_ptr, bitNum);
for(i = 0; i < bitNum; i++)
{
number |= va_arg(bits_ptr, UINT_32);
number <<= 1;
}
number >>= 1;
va_end(bits_ptr);
return number;
}
////////////////////////////////////////////////////////////////////////////////////////////////////
// Element lib
////////////////////////////////////////////////////////////////////////////////////////////////////
/**
****************************************************************************************************
* Lib::Flt32ToColorPixel
*
* @brief
* Convert a FLT_32 value to a depth/stencil pixel value
* @return
* ADDR_E_RETURNCODE
****************************************************************************************************
*/
ADDR_E_RETURNCODE Lib::Flt32ToDepthPixel(
const ELEM_FLT32TODEPTHPIXEL_INPUT* pIn,
ELEM_FLT32TODEPTHPIXEL_OUTPUT* pOut) const
{
ADDR_E_RETURNCODE returnCode = ADDR_OK;
if (GetFillSizeFieldsFlags() == TRUE)
{
if ((pIn->size != sizeof(ELEM_FLT32TODEPTHPIXEL_INPUT)) ||
(pOut->size != sizeof(ELEM_FLT32TODEPTHPIXEL_OUTPUT)))
{
returnCode = ADDR_PARAMSIZEMISMATCH;
}
}
if (returnCode == ADDR_OK)
{
GetElemLib()->Flt32ToDepthPixel(pIn->format, pIn->comps, pOut->pPixel);
UINT_32 depthBase = 0;
UINT_32 stencilBase = 0;
UINT_32 depthBits = 0;
UINT_32 stencilBits = 0;
switch (pIn->format)
{
case ADDR_DEPTH_16:
depthBits = 16;
break;
case ADDR_DEPTH_X8_24:
case ADDR_DEPTH_8_24:
case ADDR_DEPTH_X8_24_FLOAT:
case ADDR_DEPTH_8_24_FLOAT:
depthBase = 8;
depthBits = 24;
stencilBits = 8;
break;
case ADDR_DEPTH_32_FLOAT:
depthBits = 32;
break;
case ADDR_DEPTH_X24_8_32_FLOAT:
depthBase = 8;
depthBits = 32;
stencilBits = 8;
break;
default:
break;
}
// Overwrite base since R800 has no "tileBase"
if (GetElemLib()->IsDepthStencilTilePlanar() == FALSE)
{
depthBase = 0;
stencilBase = 0;
}
depthBase *= 64;
stencilBase *= 64;
pOut->stencilBase = stencilBase;
pOut->depthBase = depthBase;
pOut->depthBits = depthBits;
pOut->stencilBits = stencilBits;
}
return returnCode;
}
/**
****************************************************************************************************
* Lib::Flt32ToColorPixel
*
* @brief
* Convert a FLT_32 value to a red/green/blue/alpha pixel value
* @return
* ADDR_E_RETURNCODE
****************************************************************************************************
*/
ADDR_E_RETURNCODE Lib::Flt32ToColorPixel(
const ELEM_FLT32TOCOLORPIXEL_INPUT* pIn,
ELEM_FLT32TOCOLORPIXEL_OUTPUT* pOut) const
{
ADDR_E_RETURNCODE returnCode = ADDR_OK;
if (GetFillSizeFieldsFlags() == TRUE)
{
if ((pIn->size != sizeof(ELEM_FLT32TOCOLORPIXEL_INPUT)) ||
(pOut->size != sizeof(ELEM_FLT32TOCOLORPIXEL_OUTPUT)))
{
returnCode = ADDR_PARAMSIZEMISMATCH;
}
}
if (returnCode == ADDR_OK)
{
GetElemLib()->Flt32ToColorPixel(pIn->format,
pIn->surfNum,
pIn->surfSwap,
pIn->comps,
pOut->pPixel);
}
return returnCode;
}
/**
****************************************************************************************************
* Lib::GetExportNorm
*
* @brief
* Check one format can be EXPORT_NUM
* @return
* TRUE if EXPORT_NORM can be used
****************************************************************************************************
*/
BOOL_32 Lib::GetExportNorm(
const ELEM_GETEXPORTNORM_INPUT* pIn) const
{
ADDR_E_RETURNCODE returnCode = ADDR_OK;
BOOL_32 enabled = FALSE;
if (GetFillSizeFieldsFlags() == TRUE)
{
if (pIn->size != sizeof(ELEM_GETEXPORTNORM_INPUT))
{
returnCode = ADDR_PARAMSIZEMISMATCH;
}
}
if (returnCode == ADDR_OK)
{
enabled = GetElemLib()->PixGetExportNorm(pIn->format, pIn->num, pIn->swap);
}
return enabled;
}
/**
****************************************************************************************************
* Lib::GetBpe
*
* @brief
* Get bits-per-element for specified format
* @return
* bits-per-element of specified format
****************************************************************************************************
*/
UINT_32 Lib::GetBpe(AddrFormat format) const
{
return GetElemLib()->GetBitsPerPixel(format);
}
} // Addr
} // namespace rocr
@@ -0,0 +1,412 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
****************************************************************************************************
* @file addrlib.h
* @brief Contains the Addr::Lib base class definition.
****************************************************************************************************
*/
#ifndef __ADDR_LIB_H__
#define __ADDR_LIB_H__
#include "addrinterface.h"
#include "addrtypes.h"
#include "addrobject.h"
#include "addrelemlib.h"
#include "amdgpu_asic_addr.h"
#ifndef CIASICIDGFXENGINE_R600
#define CIASICIDGFXENGINE_R600 0x00000006
#endif
#ifndef CIASICIDGFXENGINE_R800
#define CIASICIDGFXENGINE_R800 0x00000008
#endif
#ifndef CIASICIDGFXENGINE_SOUTHERNISLAND
#define CIASICIDGFXENGINE_SOUTHERNISLAND 0x0000000A
#endif
#ifndef CIASICIDGFXENGINE_ARCTICISLAND
#define CIASICIDGFXENGINE_ARCTICISLAND 0x0000000D
#endif
namespace rocr {
namespace Addr
{
/**
****************************************************************************************************
* @brief Neutral enums that define pipeinterleave
****************************************************************************************************
*/
enum PipeInterleave
{
ADDR_PIPEINTERLEAVE_256B = 256,
ADDR_PIPEINTERLEAVE_512B = 512,
ADDR_PIPEINTERLEAVE_1KB = 1024,
ADDR_PIPEINTERLEAVE_2KB = 2048,
};
/**
****************************************************************************************************
* @brief Neutral enums that define DRAM row size
****************************************************************************************************
*/
enum RowSize
{
ADDR_ROWSIZE_1KB = 1024,
ADDR_ROWSIZE_2KB = 2048,
ADDR_ROWSIZE_4KB = 4096,
ADDR_ROWSIZE_8KB = 8192,
};
/**
****************************************************************************************************
* @brief Neutral enums that define bank interleave
****************************************************************************************************
*/
enum BankInterleave
{
ADDR_BANKINTERLEAVE_1 = 1,
ADDR_BANKINTERLEAVE_2 = 2,
ADDR_BANKINTERLEAVE_4 = 4,
ADDR_BANKINTERLEAVE_8 = 8,
};
/**
****************************************************************************************************
* @brief Neutral enums that define shader engine tile size
****************************************************************************************************
*/
enum ShaderEngineTileSize
{
ADDR_SE_TILESIZE_16 = 16,
ADDR_SE_TILESIZE_32 = 32,
};
/**
****************************************************************************************************
* @brief Neutral enums that define bank swap size
****************************************************************************************************
*/
enum BankSwapSize
{
ADDR_BANKSWAP_128B = 128,
ADDR_BANKSWAP_256B = 256,
ADDR_BANKSWAP_512B = 512,
ADDR_BANKSWAP_1KB = 1024,
};
/**
****************************************************************************************************
* @brief Enums that define max compressed fragments config
****************************************************************************************************
*/
enum NumMaxCompressedFragmentsConfig
{
ADDR_CONFIG_1_MAX_COMPRESSED_FRAGMENTS = 0x00000000,
ADDR_CONFIG_2_MAX_COMPRESSED_FRAGMENTS = 0x00000001,
ADDR_CONFIG_4_MAX_COMPRESSED_FRAGMENTS = 0x00000002,
ADDR_CONFIG_8_MAX_COMPRESSED_FRAGMENTS = 0x00000003,
};
/**
****************************************************************************************************
* @brief Enums that define num pipes config
****************************************************************************************************
*/
enum NumPipesConfig
{
ADDR_CONFIG_1_PIPE = 0x00000000,
ADDR_CONFIG_2_PIPE = 0x00000001,
ADDR_CONFIG_4_PIPE = 0x00000002,
ADDR_CONFIG_8_PIPE = 0x00000003,
ADDR_CONFIG_16_PIPE = 0x00000004,
ADDR_CONFIG_32_PIPE = 0x00000005,
ADDR_CONFIG_64_PIPE = 0x00000006,
};
/**
****************************************************************************************************
* @brief Enums that define num banks config
****************************************************************************************************
*/
enum NumBanksConfig
{
ADDR_CONFIG_1_BANK = 0x00000000,
ADDR_CONFIG_2_BANK = 0x00000001,
ADDR_CONFIG_4_BANK = 0x00000002,
ADDR_CONFIG_8_BANK = 0x00000003,
ADDR_CONFIG_16_BANK = 0x00000004,
};
/**
****************************************************************************************************
* @brief Enums that define num rb per shader engine config
****************************************************************************************************
*/
enum NumRbPerShaderEngineConfig
{
ADDR_CONFIG_1_RB_PER_SHADER_ENGINE = 0x00000000,
ADDR_CONFIG_2_RB_PER_SHADER_ENGINE = 0x00000001,
ADDR_CONFIG_4_RB_PER_SHADER_ENGINE = 0x00000002,
};
/**
****************************************************************************************************
* @brief Enums that define num shader engines config
****************************************************************************************************
*/
enum NumShaderEnginesConfig
{
ADDR_CONFIG_1_SHADER_ENGINE = 0x00000000,
ADDR_CONFIG_2_SHADER_ENGINE = 0x00000001,
ADDR_CONFIG_4_SHADER_ENGINE = 0x00000002,
ADDR_CONFIG_8_SHADER_ENGINE = 0x00000003,
};
/**
****************************************************************************************************
* @brief Enums that define pipe interleave size config
****************************************************************************************************
*/
enum PipeInterleaveSizeConfig
{
ADDR_CONFIG_PIPE_INTERLEAVE_256B = 0x00000000,
ADDR_CONFIG_PIPE_INTERLEAVE_512B = 0x00000001,
ADDR_CONFIG_PIPE_INTERLEAVE_1KB = 0x00000002,
ADDR_CONFIG_PIPE_INTERLEAVE_2KB = 0x00000003,
};
/**
****************************************************************************************************
* @brief Enums that define row size config
****************************************************************************************************
*/
enum RowSizeConfig
{
ADDR_CONFIG_1KB_ROW = 0x00000000,
ADDR_CONFIG_2KB_ROW = 0x00000001,
ADDR_CONFIG_4KB_ROW = 0x00000002,
};
/**
****************************************************************************************************
* @brief Enums that define bank interleave size config
****************************************************************************************************
*/
enum BankInterleaveSizeConfig
{
ADDR_CONFIG_BANK_INTERLEAVE_1 = 0x00000000,
ADDR_CONFIG_BANK_INTERLEAVE_2 = 0x00000001,
ADDR_CONFIG_BANK_INTERLEAVE_4 = 0x00000002,
ADDR_CONFIG_BANK_INTERLEAVE_8 = 0x00000003,
};
/**
****************************************************************************************************
* @brief Enums that define engine tile size config
****************************************************************************************************
*/
enum ShaderEngineTileSizeConfig
{
ADDR_CONFIG_SE_TILE_16 = 0x00000000,
ADDR_CONFIG_SE_TILE_32 = 0x00000001,
};
/**
****************************************************************************************************
* @brief This class contains asic independent address lib functionalities
****************************************************************************************************
*/
class Lib : public Object
{
public:
virtual ~Lib();
static ADDR_E_RETURNCODE Create(
const ADDR_CREATE_INPUT* pCreateInfo, ADDR_CREATE_OUTPUT* pCreateOut);
/// Pair of Create
VOID Destroy()
{
delete this;
}
static Lib* GetLib(ADDR_HANDLE hLib);
/// Returns AddrLib version (from compiled binary instead include file)
UINT_32 GetVersion()
{
return m_version;
}
/// Returns asic chip family name defined by AddrLib
ChipFamily GetChipFamily() const
{
return m_chipFamily;
}
ADDR_E_RETURNCODE Flt32ToDepthPixel(
const ELEM_FLT32TODEPTHPIXEL_INPUT* pIn,
ELEM_FLT32TODEPTHPIXEL_OUTPUT* pOut) const;
ADDR_E_RETURNCODE Flt32ToColorPixel(
const ELEM_FLT32TOCOLORPIXEL_INPUT* pIn,
ELEM_FLT32TOCOLORPIXEL_OUTPUT* pOut) const;
BOOL_32 GetExportNorm(const ELEM_GETEXPORTNORM_INPUT* pIn) const;
ADDR_E_RETURNCODE GetMaxAlignments(ADDR_GET_MAX_ALIGNMENTS_OUTPUT* pOut) const;
ADDR_E_RETURNCODE GetMaxMetaAlignments(ADDR_GET_MAX_ALIGNMENTS_OUTPUT* pOut) const;
UINT_32 GetBpe(AddrFormat format) const;
protected:
Lib(); // Constructor is protected
Lib(const Client* pClient);
/// Pure virtual function to get max base alignments
virtual UINT_32 HwlComputeMaxBaseAlignments() const = 0;
/// Gets maximum alignements for metadata
virtual UINT_32 HwlComputeMaxMetaBaseAlignments() const
{
ADDR_NOT_IMPLEMENTED();
return 0;
}
VOID ValidBaseAlignments(UINT_32 alignment) const
{
#if DEBUG
ADDR_ASSERT(alignment <= m_maxBaseAlign);
#endif
}
VOID ValidMetaBaseAlignments(UINT_32 metaAlignment) const
{
#if DEBUG
ADDR_ASSERT(metaAlignment <= m_maxMetaBaseAlign);
#endif
}
static BOOL_32 IsTex1d(AddrResourceType resourceType)
{
return (resourceType == ADDR_RSRC_TEX_1D);
}
static BOOL_32 IsTex2d(AddrResourceType resourceType)
{
return (resourceType == ADDR_RSRC_TEX_2D);
}
static BOOL_32 IsTex3d(AddrResourceType resourceType)
{
return (resourceType == ADDR_RSRC_TEX_3D);
}
//
// Initialization
//
/// Pure Virtual function for Hwl computing internal global parameters from h/w registers
virtual BOOL_32 HwlInitGlobalParams(const ADDR_CREATE_INPUT* pCreateIn) = 0;
/// Pure Virtual function for Hwl converting chip family
virtual ChipFamily HwlConvertChipFamily(UINT_32 uChipFamily, UINT_32 uChipRevision) = 0;
/// Get equation table pointer and number of equations
virtual UINT_32 HwlGetEquationTableInfo(const ADDR_EQUATION** ppEquationTable) const
{
*ppEquationTable = NULL;
return 0;
}
//
// Misc helper
//
static UINT_32 Bits2Number(UINT_32 bitNum, ...);
static UINT_32 GetNumFragments(UINT_32 numSamples, UINT_32 numFrags)
{
return (numFrags != 0) ? numFrags : Max(1u, numSamples);
}
/// Returns pointer of ElemLib
ElemLib* GetElemLib() const
{
return m_pElemLib;
}
/// Returns fillSizeFields flag
UINT_32 GetFillSizeFieldsFlags() const
{
return m_configFlags.fillSizeFields;
}
private:
// Disallow the copy constructor
Lib(const Lib& a);
// Disallow the assignment operator
Lib& operator=(const Lib& a);
VOID SetChipFamily(UINT_32 uChipFamily, UINT_32 uChipRevision);
VOID SetMinPitchAlignPixels(UINT_32 minPitchAlignPixels);
VOID SetMaxAlignments();
protected:
ChipFamily m_chipFamily; ///< Chip family translated from the one in atiid.h
UINT_32 m_chipRevision; ///< Revision id from xxx_id.h
UINT_32 m_version; ///< Current version
//
// Global parameters
//
ConfigFlags m_configFlags; ///< Global configuration flags. Note this is setup by
/// AddrLib instead of Client except forceLinearAligned
UINT_32 m_pipes; ///< Number of pipes
UINT_32 m_banks; ///< Number of banks
/// For r800 this is MC_ARB_RAMCFG.NOOFBANK
/// Keep it here to do default parameter calculation
UINT_32 m_pipeInterleaveBytes;
///< Specifies the size of contiguous address space
/// within each tiling pipe when making linear
/// accesses. (Formerly Group Size)
UINT_32 m_rowSize; ///< DRAM row size, in bytes
UINT_32 m_minPitchAlignPixels; ///< Minimum pitch alignment in pixels
UINT_32 m_maxSamples; ///< Max numSamples
UINT_32 m_maxBaseAlign; ///< Max base alignment for data surface
UINT_32 m_maxMetaBaseAlign; ///< Max base alignment for metadata
private:
ElemLib* m_pElemLib; ///< Element Lib pointer
};
Lib* Gfx9HwlInit (const Client* pClient);
Lib* Gfx10HwlInit(const Client* pClient);
Lib* Gfx11HwlInit(const Client* pClient);
Lib* Gfx12HwlInit(const Client* pClient);
} // Addr
} // namespace rocr
#endif
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,530 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
****************************************************************************************************
* @file addrlib1.h
* @brief Contains the Addr::V1::Lib class definition.
****************************************************************************************************
*/
#ifndef __ADDR_LIB1_H__
#define __ADDR_LIB1_H__
#include "addrlib.h"
namespace rocr {
namespace Addr
{
namespace V1
{
/**
****************************************************************************************************
* @brief Neutral enums that define bank swap size
****************************************************************************************************
*/
enum SampleSplitSize
{
ADDR_SAMPLESPLIT_1KB = 1024,
ADDR_SAMPLESPLIT_2KB = 2048,
ADDR_SAMPLESPLIT_4KB = 4096,
ADDR_SAMPLESPLIT_8KB = 8192,
};
/**
****************************************************************************************************
* @brief Flags for AddrTileMode
****************************************************************************************************
*/
struct TileModeFlags
{
UINT_32 thickness : 4;
UINT_32 isLinear : 1;
UINT_32 isMicro : 1;
UINT_32 isMacro : 1;
UINT_32 isMacro3d : 1;
UINT_32 isPrt : 1;
UINT_32 isPrtNoRotation : 1;
UINT_32 isBankSwapped : 1;
};
static const UINT_32 Block64K = 0x10000;
static const UINT_32 PrtTileSize = Block64K;
/**
****************************************************************************************************
* @brief This class contains asic independent address lib functionalities
****************************************************************************************************
*/
class Lib : public Addr::Lib
{
public:
virtual ~Lib();
static Lib* GetLib(
ADDR_HANDLE hLib);
/// Returns tileIndex support
BOOL_32 UseTileIndex(INT_32 index) const
{
return m_configFlags.useTileIndex && (index != TileIndexInvalid);
}
/// Returns combined swizzle support
BOOL_32 UseCombinedSwizzle() const
{
return m_configFlags.useCombinedSwizzle;
}
//
// Interface stubs
//
ADDR_E_RETURNCODE ComputeSurfaceInfo(
const ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoord(
const ADDR_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceCoordFromAddr(
const ADDR_COMPUTE_SURFACE_COORDFROMADDR_INPUT* pIn,
ADDR_COMPUTE_SURFACE_COORDFROMADDR_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSliceTileSwizzle(
const ADDR_COMPUTE_SLICESWIZZLE_INPUT* pIn,
ADDR_COMPUTE_SLICESWIZZLE_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ExtractBankPipeSwizzle(
const ADDR_EXTRACT_BANKPIPE_SWIZZLE_INPUT* pIn,
ADDR_EXTRACT_BANKPIPE_SWIZZLE_OUTPUT* pOut) const;
ADDR_E_RETURNCODE CombineBankPipeSwizzle(
const ADDR_COMBINE_BANKPIPE_SWIZZLE_INPUT* pIn,
ADDR_COMBINE_BANKPIPE_SWIZZLE_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeBaseSwizzle(
const ADDR_COMPUTE_BASE_SWIZZLE_INPUT* pIn,
ADDR_COMPUTE_BASE_SWIZZLE_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeFmaskInfo(
const ADDR_COMPUTE_FMASK_INFO_INPUT* pIn,
ADDR_COMPUTE_FMASK_INFO_OUTPUT* pOut);
ADDR_E_RETURNCODE ComputeFmaskAddrFromCoord(
const ADDR_COMPUTE_FMASK_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_FMASK_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeFmaskCoordFromAddr(
const ADDR_COMPUTE_FMASK_COORDFROMADDR_INPUT* pIn,
ADDR_COMPUTE_FMASK_COORDFROMADDR_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ConvertTileInfoToHW(
const ADDR_CONVERT_TILEINFOTOHW_INPUT* pIn,
ADDR_CONVERT_TILEINFOTOHW_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ConvertTileIndex(
const ADDR_CONVERT_TILEINDEX_INPUT* pIn,
ADDR_CONVERT_TILEINDEX_OUTPUT* pOut) const;
ADDR_E_RETURNCODE GetMacroModeIndex(
const ADDR_GET_MACROMODEINDEX_INPUT* pIn,
ADDR_GET_MACROMODEINDEX_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ConvertTileIndex1(
const ADDR_CONVERT_TILEINDEX1_INPUT* pIn,
ADDR_CONVERT_TILEINDEX_OUTPUT* pOut) const;
ADDR_E_RETURNCODE GetTileIndex(
const ADDR_GET_TILEINDEX_INPUT* pIn,
ADDR_GET_TILEINDEX_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeHtileInfo(
const ADDR_COMPUTE_HTILE_INFO_INPUT* pIn,
ADDR_COMPUTE_HTILE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeCmaskInfo(
const ADDR_COMPUTE_CMASK_INFO_INPUT* pIn,
ADDR_COMPUTE_CMASK_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeDccInfo(
const ADDR_COMPUTE_DCCINFO_INPUT* pIn,
ADDR_COMPUTE_DCCINFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeHtileAddrFromCoord(
const ADDR_COMPUTE_HTILE_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_HTILE_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeCmaskAddrFromCoord(
const ADDR_COMPUTE_CMASK_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_CMASK_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeHtileCoordFromAddr(
const ADDR_COMPUTE_HTILE_COORDFROMADDR_INPUT* pIn,
ADDR_COMPUTE_HTILE_COORDFROMADDR_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeCmaskCoordFromAddr(
const ADDR_COMPUTE_CMASK_COORDFROMADDR_INPUT* pIn,
ADDR_COMPUTE_CMASK_COORDFROMADDR_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputePrtInfo(
const ADDR_PRT_INFO_INPUT* pIn,
ADDR_PRT_INFO_OUTPUT* pOut) const;
protected:
Lib(); // Constructor is protected
Lib(const Client* pClient);
/// Pure Virtual function for Hwl computing surface info
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfo(
const ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl computing surface address from coord
virtual ADDR_E_RETURNCODE HwlComputeSurfaceAddrFromCoord(
const ADDR_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl computing surface coord from address
virtual ADDR_E_RETURNCODE HwlComputeSurfaceCoordFromAddr(
const ADDR_COMPUTE_SURFACE_COORDFROMADDR_INPUT* pIn,
ADDR_COMPUTE_SURFACE_COORDFROMADDR_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl computing surface tile swizzle
virtual ADDR_E_RETURNCODE HwlComputeSliceTileSwizzle(
const ADDR_COMPUTE_SLICESWIZZLE_INPUT* pIn,
ADDR_COMPUTE_SLICESWIZZLE_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl extracting bank/pipe swizzle from base256b
virtual ADDR_E_RETURNCODE HwlExtractBankPipeSwizzle(
const ADDR_EXTRACT_BANKPIPE_SWIZZLE_INPUT* pIn,
ADDR_EXTRACT_BANKPIPE_SWIZZLE_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl combining bank/pipe swizzle
virtual ADDR_E_RETURNCODE HwlCombineBankPipeSwizzle(
UINT_32 bankSwizzle, UINT_32 pipeSwizzle, ADDR_TILEINFO* pTileInfo,
UINT_64 baseAddr, UINT_32* pTileSwizzle) const = 0;
/// Pure Virtual function for Hwl computing base swizzle
virtual ADDR_E_RETURNCODE HwlComputeBaseSwizzle(
const ADDR_COMPUTE_BASE_SWIZZLE_INPUT* pIn,
ADDR_COMPUTE_BASE_SWIZZLE_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl computing HTILE base align
virtual UINT_32 HwlComputeHtileBaseAlign(
BOOL_32 isTcCompatible, BOOL_32 isLinear, ADDR_TILEINFO* pTileInfo) const = 0;
/// Pure Virtual function for Hwl computing HTILE bpp
virtual UINT_32 HwlComputeHtileBpp(
BOOL_32 isWidth8, BOOL_32 isHeight8) const = 0;
/// Pure Virtual function for Hwl computing HTILE bytes
virtual UINT_64 HwlComputeHtileBytes(
UINT_32 pitch, UINT_32 height, UINT_32 bpp,
BOOL_32 isLinear, UINT_32 numSlices, UINT_64* pSliceBytes, UINT_32 baseAlign) const = 0;
/// Pure Virtual function for Hwl computing FMASK info
virtual ADDR_E_RETURNCODE HwlComputeFmaskInfo(
const ADDR_COMPUTE_FMASK_INFO_INPUT* pIn,
ADDR_COMPUTE_FMASK_INFO_OUTPUT* pOut) = 0;
/// Pure Virtual function for Hwl FMASK address from coord
virtual ADDR_E_RETURNCODE HwlComputeFmaskAddrFromCoord(
const ADDR_COMPUTE_FMASK_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_FMASK_ADDRFROMCOORD_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl FMASK coord from address
virtual ADDR_E_RETURNCODE HwlComputeFmaskCoordFromAddr(
const ADDR_COMPUTE_FMASK_COORDFROMADDR_INPUT* pIn,
ADDR_COMPUTE_FMASK_COORDFROMADDR_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl convert tile info from real value to HW value
virtual ADDR_E_RETURNCODE HwlConvertTileInfoToHW(
const ADDR_CONVERT_TILEINFOTOHW_INPUT* pIn,
ADDR_CONVERT_TILEINFOTOHW_OUTPUT* pOut) const = 0;
/// Pure Virtual function for Hwl compute mipmap info
virtual BOOL_32 HwlComputeMipLevel(
ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn) const = 0;
/// Pure Virtual function for Hwl compute max cmask blockMax value
virtual BOOL_32 HwlGetMaxCmaskBlockMax() const = 0;
/// Pure Virtual function for Hwl compute fmask bits
virtual UINT_32 HwlComputeFmaskBits(
const ADDR_COMPUTE_FMASK_INFO_INPUT* pIn,
UINT_32* pNumSamples) const = 0;
/// Virtual function to get index (not pure then no need to implement this in all hwls
virtual ADDR_E_RETURNCODE HwlGetTileIndex(
const ADDR_GET_TILEINDEX_INPUT* pIn,
ADDR_GET_TILEINDEX_OUTPUT* pOut) const
{
return ADDR_NOTSUPPORTED;
}
/// Virtual function for Hwl to compute Dcc info
virtual ADDR_E_RETURNCODE HwlComputeDccInfo(
const ADDR_COMPUTE_DCCINFO_INPUT* pIn,
ADDR_COMPUTE_DCCINFO_OUTPUT* pOut) const
{
return ADDR_NOTSUPPORTED;
}
/// Virtual function to get cmask address for tc compatible cmask
virtual ADDR_E_RETURNCODE HwlComputeCmaskAddrFromCoord(
const ADDR_COMPUTE_CMASK_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_CMASK_ADDRFROMCOORD_OUTPUT* pOut) const
{
return ADDR_NOTSUPPORTED;
}
/// Virtual function to get htile address for tc compatible htile
virtual ADDR_E_RETURNCODE HwlComputeHtileAddrFromCoord(
const ADDR_COMPUTE_HTILE_ADDRFROMCOORD_INPUT* pIn,
ADDR_COMPUTE_HTILE_ADDRFROMCOORD_OUTPUT* pOut) const
{
return ADDR_NOTSUPPORTED;
}
// Compute attributes
// HTILE
UINT_32 ComputeHtileInfo(
ADDR_HTILE_FLAGS flags,
UINT_32 pitchIn, UINT_32 heightIn, UINT_32 numSlices,
BOOL_32 isLinear, BOOL_32 isWidth8, BOOL_32 isHeight8,
ADDR_TILEINFO* pTileInfo,
UINT_32* pPitchOut, UINT_32* pHeightOut, UINT_64* pHtileBytes,
UINT_32* pMacroWidth = NULL, UINT_32* pMacroHeight = NULL,
UINT_64* pSliceSize = NULL, UINT_32* pBaseAlign = NULL) const;
// CMASK
ADDR_E_RETURNCODE ComputeCmaskInfo(
ADDR_CMASK_FLAGS flags,
UINT_32 pitchIn, UINT_32 heightIn, UINT_32 numSlices, BOOL_32 isLinear,
ADDR_TILEINFO* pTileInfo, UINT_32* pPitchOut, UINT_32* pHeightOut, UINT_64* pCmaskBytes,
UINT_32* pMacroWidth, UINT_32* pMacroHeight, UINT_64* pSliceSize = NULL,
UINT_32* pBaseAlign = NULL, UINT_32* pBlockMax = NULL) const;
virtual VOID HwlComputeTileDataWidthAndHeightLinear(
UINT_32* pMacroWidth, UINT_32* pMacroHeight,
UINT_32 bpp, ADDR_TILEINFO* pTileInfo) const;
// CMASK & HTILE addressing
virtual UINT_64 HwlComputeXmaskAddrFromCoord(
UINT_32 pitch, UINT_32 height, UINT_32 x, UINT_32 y, UINT_32 slice,
UINT_32 numSlices, UINT_32 factor, BOOL_32 isLinear, BOOL_32 isWidth8,
BOOL_32 isHeight8, ADDR_TILEINFO* pTileInfo,
UINT_32* bitPosition) const;
virtual VOID HwlComputeXmaskCoordFromAddr(
UINT_64 addr, UINT_32 bitPosition, UINT_32 pitch, UINT_32 height, UINT_32 numSlices,
UINT_32 factor, BOOL_32 isLinear, BOOL_32 isWidth8, BOOL_32 isHeight8,
ADDR_TILEINFO* pTileInfo, UINT_32* pX, UINT_32* pY, UINT_32* pSlice) const;
// Surface mipmap
VOID ComputeMipLevel(
ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
/// Pure Virtual function for Hwl to get macro tiled alignment info
virtual BOOL_32 HwlGetAlignmentInfoMacroTiled(
const ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn,
UINT_32* pPitchAlign, UINT_32* pHeightAlign, UINT_32* pSizeAlign) const = 0;
virtual VOID HwlOverrideTileMode(ADDR_COMPUTE_SURFACE_INFO_INPUT* pInOut) const
{
// not supported in hwl layer
}
virtual VOID HwlOptimizeTileMode(ADDR_COMPUTE_SURFACE_INFO_INPUT* pInOut) const
{
// not supported in hwl layer
}
virtual VOID HwlSelectTileMode(ADDR_COMPUTE_SURFACE_INFO_INPUT* pInOut) const
{
// not supported in hwl layer
}
AddrTileMode DegradeLargeThickTile(AddrTileMode tileMode, UINT_32 bpp) const;
VOID PadDimensions(
AddrTileMode tileMode, UINT_32 bpp, ADDR_SURFACE_FLAGS flags,
UINT_32 numSamples, ADDR_TILEINFO* pTileInfo, UINT_32 padDims, UINT_32 mipLevel,
UINT_32* pPitch, UINT_32* pPitchAlign, UINT_32* pHeight, UINT_32 heightAlign,
UINT_32* pSlices, UINT_32 sliceAlign) const;
virtual VOID HwlPadDimensions(
AddrTileMode tileMode, UINT_32 bpp, ADDR_SURFACE_FLAGS flags,
UINT_32 numSamples, ADDR_TILEINFO* pTileInfo, UINT_32 mipLevel,
UINT_32* pPitch, UINT_32* pPitchAlign, UINT_32 height, UINT_32 heightAlign) const
{
}
//
// Addressing shared for linear/1D tiling
//
UINT_64 ComputeSurfaceAddrFromCoordLinear(
UINT_32 x, UINT_32 y, UINT_32 slice, UINT_32 sample,
UINT_32 bpp, UINT_32 pitch, UINT_32 height, UINT_32 numSlices,
UINT_32* pBitPosition) const;
VOID ComputeSurfaceCoordFromAddrLinear(
UINT_64 addr, UINT_32 bitPosition, UINT_32 bpp,
UINT_32 pitch, UINT_32 height, UINT_32 numSlices,
UINT_32* pX, UINT_32* pY, UINT_32* pSlice, UINT_32* pSample) const;
VOID ComputeSurfaceCoordFromAddrMicroTiled(
UINT_64 addr, UINT_32 bitPosition,
UINT_32 bpp, UINT_32 pitch, UINT_32 height, UINT_32 numSamples,
AddrTileMode tileMode, UINT_32 tileBase, UINT_32 compBits,
UINT_32* pX, UINT_32* pY, UINT_32* pSlice, UINT_32* pSample,
AddrTileType microTileType, BOOL_32 isDepthSampleOrder) const;
ADDR_E_RETURNCODE ComputeMicroTileEquation(
UINT_32 bpp, AddrTileMode tileMode,
AddrTileType microTileType, ADDR_EQUATION* pEquation) const;
UINT_32 ComputePixelIndexWithinMicroTile(
UINT_32 x, UINT_32 y, UINT_32 z,
UINT_32 bpp, AddrTileMode tileMode, AddrTileType microTileType) const;
/// Pure Virtual function for Hwl computing coord from offset inside micro tile
virtual VOID HwlComputePixelCoordFromOffset(
UINT_32 offset, UINT_32 bpp, UINT_32 numSamples,
AddrTileMode tileMode, UINT_32 tileBase, UINT_32 compBits,
UINT_32* pX, UINT_32* pY, UINT_32* pSlice, UINT_32* pSample,
AddrTileType microTileType, BOOL_32 isDepthSampleOrder) const = 0;
//
// Addressing shared by all
//
virtual UINT_32 HwlGetPipes(
const ADDR_TILEINFO* pTileInfo) const;
UINT_32 ComputePipeFromAddr(
UINT_64 addr, UINT_32 numPipes) const;
virtual ADDR_E_RETURNCODE ComputePipeEquation(
UINT_32 log2BytesPP, UINT_32 threshX, UINT_32 threshY, ADDR_TILEINFO* pTileInfo, ADDR_EQUATION* pEquation) const
{
return ADDR_NOTSUPPORTED;
}
/// Pure Virtual function for Hwl computing pipe from coord
virtual UINT_32 ComputePipeFromCoord(
UINT_32 x, UINT_32 y, UINT_32 slice, AddrTileMode tileMode,
UINT_32 pipeSwizzle, BOOL_32 flags, ADDR_TILEINFO* pTileInfo) const = 0;
/// Pure Virtual function for Hwl computing coord Y for 8 pipe cmask/htile
virtual UINT_32 HwlComputeXmaskCoordYFrom8Pipe(
UINT_32 pipe, UINT_32 x) const = 0;
//
// Misc helper
//
static const TileModeFlags ModeFlags[ADDR_TM_COUNT];
static UINT_32 Thickness(
AddrTileMode tileMode);
// Checking tile mode
static BOOL_32 IsMacroTiled(AddrTileMode tileMode);
static BOOL_32 IsMacro3dTiled(AddrTileMode tileMode);
static BOOL_32 IsLinear(AddrTileMode tileMode);
static BOOL_32 IsMicroTiled(AddrTileMode tileMode);
static BOOL_32 IsPrtTileMode(AddrTileMode tileMode);
static BOOL_32 IsPrtNoRotationTileMode(AddrTileMode tileMode);
/// Return TRUE if tile info is needed
BOOL_32 UseTileInfo() const
{
return !m_configFlags.ignoreTileInfo;
}
/// Adjusts pitch alignment for flipping surface
VOID AdjustPitchAlignment(
ADDR_SURFACE_FLAGS flags, UINT_32* pPitchAlign) const;
/// Overwrite tile config according to tile index
virtual ADDR_E_RETURNCODE HwlSetupTileCfg(
UINT_32 bpp, INT_32 index, INT_32 macroModeIndex,
ADDR_TILEINFO* pInfo, AddrTileMode* mode = NULL, AddrTileType* type = NULL) const;
/// Overwrite macro tile config according to tile index
virtual INT_32 HwlComputeMacroModeIndex(
INT_32 index, ADDR_SURFACE_FLAGS flags, UINT_32 bpp, UINT_32 numSamples,
ADDR_TILEINFO* pTileInfo, AddrTileMode *pTileMode = NULL, AddrTileType *pTileType = NULL
) const
{
return TileIndexNoMacroIndex;
}
/// Pre-handler of 3x pitch (96 bit) adjustment
virtual UINT_32 HwlPreHandleBaseLvl3xPitch(
const ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn, UINT_32 expPitch) const;
/// Post-handler of 3x pitch adjustment
virtual UINT_32 HwlPostHandleBaseLvl3xPitch(
const ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn, UINT_32 expPitch) const;
/// Check miplevel after surface adjustment
ADDR_E_RETURNCODE PostComputeMipLevel(
ADDR_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
/// Quad buffer stereo support, has its implementation in ind. layer
VOID ComputeQbStereoInfo(
ADDR_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
/// Pure virutual function to compute stereo bank swizzle for right eye
virtual UINT_32 HwlComputeQbStereoRightSwizzle(
ADDR_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const = 0;
VOID OptimizeTileMode(ADDR_COMPUTE_SURFACE_INFO_INPUT* pInOut) const;
/// Overwrite tile setting to PRT
virtual VOID HwlSetPrtTileMode(ADDR_COMPUTE_SURFACE_INFO_INPUT* pInOut) const
{
}
static BOOL_32 DegradeTo1D(
UINT_32 width, UINT_32 height,
UINT_32 macroTilePitchAlign, UINT_32 macroTileHeightAlign);
private:
// Disallow the copy constructor
Lib(const Lib& a);
// Disallow the assignment operator
Lib& operator=(const Lib& a);
UINT_32 ComputeCmaskBaseAlign(
ADDR_CMASK_FLAGS flags, ADDR_TILEINFO* pTileInfo) const;
UINT_64 ComputeCmaskBytes(
UINT_32 pitch, UINT_32 height, UINT_32 numSlices) const;
//
// CMASK/HTILE shared methods
//
VOID ComputeTileDataWidthAndHeight(
UINT_32 bpp, UINT_32 cacheBits, ADDR_TILEINFO* pTileInfo,
UINT_32* pMacroWidth, UINT_32* pMacroHeight) const;
UINT_32 ComputeXmaskCoordYFromPipe(
UINT_32 pipe, UINT_32 x) const;
};
} // V1
} // Addr
} // namespace rocr
#endif
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,417 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2023 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
************************************************************************************************************************
* @file addrlib3.h
* @brief Contains the Addr::V3::Lib class definition.
************************************************************************************************************************
*/
#ifndef __ADDR3_LIB3_H__
#define __ADDR3_LIB3_H__
#include "addrlib.h"
namespace rocr {
namespace Addr
{
namespace V3
{
/**
************************************************************************************************************************
* @brief Bitmasks for swizzle mode determination on GFX12
************************************************************************************************************************
*/
const UINT_32 Gfx12Blk256KBSwModeMask = (1u << ADDR3_256KB_2D) |
(1u << ADDR3_256KB_3D);
const UINT_32 Gfx12Blk64KBSwModeMask = (1u << ADDR3_64KB_2D) |
(1u << ADDR3_64KB_3D);
const UINT_32 Gfx12Blk4KBSwModeMask = (1u << ADDR3_4KB_2D) |
(1u << ADDR3_4KB_3D);
const UINT_32 Gfx12Blk256BSwModeMask = (1u << ADDR3_256B_2D);
/**
************************************************************************************************************************
* @brief Bit setting for swizzle pattern
************************************************************************************************************************
*/
union ADDR_BIT_SETTING
{
struct
{
UINT_16 x;
UINT_16 y;
UINT_16 z;
UINT_16 s;
};
UINT_64 value;
};
/**
************************************************************************************************************************
* @brief Flags for SwizzleModeTable
************************************************************************************************************************
*/
union SwizzleModeFlags
{
struct
{
// Swizzle mode
UINT_32 isLinear : 1; // Linear
UINT_32 is2d : 1; // 2d mode
UINT_32 is3d : 1; // 3d mode
// Block size
UINT_32 is256b : 1; // Block size is 256B
UINT_32 is4kb : 1; // Block size is 4KB
UINT_32 is64kb : 1; // Block size is 64KB
UINT_32 is256kb : 1; // Block size is 256KB
UINT_32 reserved : 25; // Reserved bits
};
UINT_32 u32All;
};
struct Dim2d
{
UINT_32 w;
UINT_32 h;
};
const UINT_32 Log2Size256 = 8u;
const UINT_32 Log2Size4K = 12u;
const UINT_32 Log2Size64K = 16u;
const UINT_32 Log2Size256K = 18u;
/**
************************************************************************************************************************
* @brief Swizzle pattern information
************************************************************************************************************************
*/
// Accessed by index representing the logbase2 of (8bpp/16bpp/32bpp/64bpp/128bpp)
// contains the indices which map to 2D arrays SW_PATTERN_NIBBLE[1-4] which contain sections of an index equation.
struct ADDR_SW_PATINFO
{
UINT_8 nibble1Idx;
UINT_8 nibble2Idx;
UINT_8 nibble3Idx;
UINT_8 nibble4Idx;
};
/**
************************************************************************************************************************
* InitBit
*
* @brief
* Initialize bit setting value via a return value
************************************************************************************************************************
*/
#define InitBit(c, index) (1ull << ((c << 4) + index))
const UINT_64 X0 = InitBit(0, 0);
const UINT_64 X1 = InitBit(0, 1);
const UINT_64 X2 = InitBit(0, 2);
const UINT_64 X3 = InitBit(0, 3);
const UINT_64 X4 = InitBit(0, 4);
const UINT_64 X5 = InitBit(0, 5);
const UINT_64 X6 = InitBit(0, 6);
const UINT_64 X7 = InitBit(0, 7);
const UINT_64 X8 = InitBit(0, 8);
const UINT_64 Y0 = InitBit(1, 0);
const UINT_64 Y1 = InitBit(1, 1);
const UINT_64 Y2 = InitBit(1, 2);
const UINT_64 Y3 = InitBit(1, 3);
const UINT_64 Y4 = InitBit(1, 4);
const UINT_64 Y5 = InitBit(1, 5);
const UINT_64 Y6 = InitBit(1, 6);
const UINT_64 Y7 = InitBit(1, 7);
const UINT_64 Y8 = InitBit(1, 8);
const UINT_64 Z0 = InitBit(2, 0);
const UINT_64 Z1 = InitBit(2, 1);
const UINT_64 Z2 = InitBit(2, 2);
const UINT_64 Z3 = InitBit(2, 3);
const UINT_64 Z4 = InitBit(2, 4);
const UINT_64 Z5 = InitBit(2, 5);
const UINT_64 S0 = InitBit(3, 0);
const UINT_64 S1 = InitBit(3, 1);
const UINT_64 S2 = InitBit(3, 2);
/**
************************************************************************************************************************
* @brief Bit setting for swizzle pattern
************************************************************************************************************************
*/
/**
************************************************************************************************************************
* @brief This class contains asic independent address lib functionalities
************************************************************************************************************************
*/
class Lib : public Addr::Lib
{
public:
virtual ~Lib();
static Lib* GetLib(
ADDR_HANDLE hLib);
//
// Interface stubs
//
// For data surface
ADDR_E_RETURNCODE ComputeSurfaceInfo(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE GetPossibleSwizzleModes(
const ADDR3_GET_POSSIBLE_SWIZZLE_MODE_INPUT* pIn,
ADDR3_GET_POSSIBLE_SWIZZLE_MODE_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoord(
const ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
// Misc
ADDR_E_RETURNCODE ComputePipeBankXor(
const ADDR3_COMPUTE_PIPEBANKXOR_INPUT* pIn,
ADDR3_COMPUTE_PIPEBANKXOR_OUTPUT* pOut);
ADDR_E_RETURNCODE ComputeNonBlockCompressedView(
const ADDR3_COMPUTE_NONBLOCKCOMPRESSEDVIEW_INPUT* pIn,
ADDR3_COMPUTE_NONBLOCKCOMPRESSEDVIEW_OUTPUT* pOut);
ADDR_E_RETURNCODE ComputeSubResourceOffsetForSwizzlePattern(
const ADDR3_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_INPUT* pIn,
ADDR3_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_OUTPUT* pOut);
ADDR_E_RETURNCODE ComputeSlicePipeBankXor(
const ADDR3_COMPUTE_SLICE_PIPEBANKXOR_INPUT* pIn,
ADDR3_COMPUTE_SLICE_PIPEBANKXOR_OUTPUT* pOut);
protected:
Lib(); // Constructor is protected
Lib(const Client* pClient);
static const UINT_32 MaxImageDim = 65536;
static const UINT_32 MaxMipLevels = 17; // Max image size is 64k
static const UINT_32 MaxNumOfBpp = 5;
static const UINT_32 MaxNumOfAA = 4;
UINT_32 m_pipesLog2; ///< Number of pipe per shader engine Log2
UINT_32 m_pipeInterleaveLog2; ///< Log2 of pipe interleave bytes
static const Dim2d Block256_2d[MaxNumOfBpp];
static const ADDR_EXTENT3D Block1K_3d[MaxNumOfBpp];
SwizzleModeFlags m_swizzleModeTable[ADDR3_MAX_TYPE]; ///< Swizzle mode table
// Number of unique MSAA sample rates (1/2/4/8)
static const UINT_32 MaxMsaaRateLog2 = 4;
// Max number of bpp (8bpp/16bpp/32bpp/64bpp/128bpp)
static const UINT_32 MaxElementBytesLog2 = 5;
// Number of unique swizzle patterns (one entry per swizzle mode + MSAA + bpp configuration)
static const UINT_32 NumSwizzlePatterns = 19 * MaxElementBytesLog2;
// Number of equation entries in the table
UINT_32 m_numEquations;
// Equation lookup table according to swizzle mode, MSAA sample rate, and bpp
UINT_32 m_equationLookupTable[ADDR3_MAX_TYPE - 1][MaxMsaaRateLog2][MaxElementBytesLog2];
// Equation table
ADDR_EQUATION m_equationTable[NumSwizzlePatterns];
void SetEquationTableEntry(
Addr3SwizzleMode addrType,
UINT_32 msaaLog2,
UINT_32 elementLog2,
UINT_32 value)
{
m_equationLookupTable[addrType - 1][msaaLog2][elementLog2] = value;
}
const UINT_32 GetEquationTableEntry(
Addr3SwizzleMode addrType,
UINT_32 msaaLog2,
UINT_32 elementLog2) const
{
return m_equationLookupTable[addrType - 1][msaaLog2][elementLog2];
}
static BOOL_32 Valid3DMipSliceIdConstraint(
UINT_32 numSlices,
UINT_32 mipId,
UINT_32 slice)
{
return (Max((numSlices >> mipId), 1u) > slice);
}
UINT_32 GetBlockSize(
Addr3SwizzleMode swizzleMode,
BOOL_32 forPitch = FALSE) const;
UINT_32 GetBlockSizeLog2(
Addr3SwizzleMode swizzleMode,
BOOL_32 forPitch = FALSE) const;
BOOL_32 IsValidSwMode(Addr3SwizzleMode swizzleMode) const
{
return (m_swizzleModeTable[swizzleMode].u32All != 0);
}
UINT_32 IsLinear(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].isLinear;
}
// Checking block size
BOOL_32 IsBlock256b(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].is256b;
}
// Checking block size
BOOL_32 IsBlock4kb(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].is4kb;
}
// Checking block size
BOOL_32 IsBlock64kb(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].is64kb;
}
// Checking block size
BOOL_32 IsBlock256kb(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].is256kb;
}
BOOL_32 Is2dSwizzle(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].is2d;
}
BOOL_32 Is3dSwizzle(Addr3SwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].is3d;
}
virtual UINT_32 HwlComputeMaxBaseAlignments() const { return 256 * 1024; }
virtual BOOL_32 HwlInitGlobalParams(const ADDR_CREATE_INPUT* pCreateIn)
{
ADDR_NOT_IMPLEMENTED();
// Although GFX12 addressing should be consistent regardless of the configuration, we still need to
// call some initialization for member variables.
return TRUE;
}
virtual ChipFamily HwlConvertChipFamily(
UINT_32 chipFamily,
UINT_32 chipRevision);
virtual UINT_32 HwlComputeMaxMetaBaseAlignments() const { return 0; }
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfo(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const
{
ADDR_NOT_IMPLEMENTED();
return ADDR_NOTSUPPORTED;
}
virtual ADDR_E_RETURNCODE HwlComputePipeBankXor(
const ADDR3_COMPUTE_PIPEBANKXOR_INPUT* pIn,
ADDR3_COMPUTE_PIPEBANKXOR_OUTPUT* pOut) const
{
ADDR_NOT_IMPLEMENTED();
return ADDR_NOTSUPPORTED;
}
VOID ComputeBlockDimensionForSurf(
ADDR_EXTENT3D* pExtent,
UINT_32 bpp,
UINT_32 numSamples,
Addr3SwizzleMode swizzleMode) const;
ADDR_EXTENT3D GetMipTailDim(
Addr3SwizzleMode swizzleMode,
const ADDR_EXTENT3D& blockDims) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoordLinear(
const ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoordTiled(
const ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceAddrFromCoordTiled(
const ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const
{
ADDR_NOT_IMPLEMENTED();
return ADDR_NOTIMPLEMENTED;
}
virtual ADDR_E_RETURNCODE HwlComputeNonBlockCompressedView(
const ADDR3_COMPUTE_NONBLOCKCOMPRESSEDVIEW_INPUT* pIn,
ADDR3_COMPUTE_NONBLOCKCOMPRESSEDVIEW_OUTPUT* pOut) const
{
ADDR_NOT_IMPLEMENTED();
return ADDR_NOTSUPPORTED;
}
virtual VOID HwlComputeSubResourceOffsetForSwizzlePattern(
const ADDR3_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_INPUT* pIn,
ADDR3_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_OUTPUT* pOut) const
{
ADDR_NOT_IMPLEMENTED();
}
virtual ADDR_E_RETURNCODE HwlComputeSlicePipeBankXor(
const ADDR3_COMPUTE_SLICE_PIPEBANKXOR_INPUT* pIn,
ADDR3_COMPUTE_SLICE_PIPEBANKXOR_OUTPUT* pOut) const
{
ADDR_NOT_IMPLEMENTED();
return ADDR_NOTSUPPORTED;
}
ADDR_E_RETURNCODE ApplyCustomizedPitchHeight(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
BOOL_32 UseCustomHeight(const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
BOOL_32 UseCustomPitch(const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
BOOL_32 CanTrimLinearPadding(const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
private:
// Disallow the copy constructor
Lib(const Lib& a);
// Disallow the assignment operator
Lib& operator=(const Lib& a);
void Init();
};
} // V3
} // Addr
} // namespace rocr
#endif
@@ -0,0 +1,224 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
****************************************************************************************************
* @file addrobject.cpp
* @brief Contains the Object base class implementation.
****************************************************************************************************
*/
#include "addrinterface.h"
#include "addrobject.h"
namespace rocr {
namespace Addr
{
/**
****************************************************************************************************
* Object::Object
*
* @brief
* Constructor for the Object class.
****************************************************************************************************
*/
Object::Object()
{
m_client.handle = NULL;
m_client.callbacks.allocSysMem = NULL;
m_client.callbacks.freeSysMem = NULL;
m_client.callbacks.debugPrint = NULL;
}
/**
****************************************************************************************************
* Object::Object
*
* @brief
* Constructor for the Object class.
****************************************************************************************************
*/
Object::Object(const Client* pClient)
{
m_client = *pClient;
}
/**
****************************************************************************************************
* Object::~Object
*
* @brief
* Destructor for the Object class.
****************************************************************************************************
*/
Object::~Object()
{
}
/**
****************************************************************************************************
* Object::ClientAlloc
*
* @brief
* Calls instanced allocSysMem inside Client
****************************************************************************************************
*/
VOID* Object::ClientAlloc(
size_t objSize, ///< [in] Size to allocate
const Client* pClient) ///< [in] Client pointer
{
VOID* pObjMem = NULL;
if (pClient->callbacks.allocSysMem != NULL)
{
ADDR_ALLOCSYSMEM_INPUT allocInput = {0};
allocInput.size = sizeof(ADDR_ALLOCSYSMEM_INPUT);
allocInput.flags.value = 0;
allocInput.sizeInBytes = static_cast<UINT_32>(objSize);
allocInput.hClient = pClient->handle;
pObjMem = pClient->callbacks.allocSysMem(&allocInput);
}
return pObjMem;
}
/**
****************************************************************************************************
* Object::Alloc
*
* @brief
* A wrapper of ClientAlloc
****************************************************************************************************
*/
VOID* Object::Alloc(
size_t objSize ///< [in] Size to allocate
) const
{
return ClientAlloc(objSize, &m_client);;
}
/**
****************************************************************************************************
* Object::ClientFree
*
* @brief
* Calls freeSysMem inside Client
****************************************************************************************************
*/
VOID Object::ClientFree(
VOID* pObjMem, ///< [in] User virtual address to free.
const Client* pClient) ///< [in] Client pointer
{
if (pClient->callbacks.freeSysMem != NULL)
{
if (pObjMem != NULL)
{
ADDR_FREESYSMEM_INPUT freeInput = {0};
freeInput.size = sizeof(ADDR_FREESYSMEM_INPUT);
freeInput.hClient = pClient->handle;
freeInput.pVirtAddr = pObjMem;
pClient->callbacks.freeSysMem(&freeInput);
}
}
}
/**
****************************************************************************************************
* Object::Free
*
* @brief
* A wrapper of ClientFree
****************************************************************************************************
*/
VOID Object::Free(
VOID* pObjMem ///< [in] User virtual address to free.
) const
{
ClientFree(pObjMem, &m_client);
}
/**
****************************************************************************************************
* Object::operator new
*
* @brief
* Placement new operator. (with pre-allocated memory pointer)
*
* @return
* Returns pre-allocated memory pointer.
****************************************************************************************************
*/
VOID* Object::operator new(
size_t objSize, ///< [in] Size to allocate
VOID* pMem ///< [in] Pre-allocated pointer
) noexcept
{
return pMem;
}
/**
****************************************************************************************************
* Object::operator delete
*
* @brief
* Frees Object object memory.
****************************************************************************************************
*/
VOID Object::operator delete(
VOID* pObjMem) ///< [in] User virtual address to free.
{
Object* pObj = static_cast<Object*>(pObjMem);
ClientFree(pObjMem, &pObj->m_client);
}
/**
****************************************************************************************************
* Object::DebugPrint
*
* @brief
* Print debug message
*
* @return
* N/A
****************************************************************************************************
*/
VOID Object::DebugPrint(
const CHAR* pDebugString, ///< [in] Debug string
...
) const
{
#if DEBUG
if (m_client.callbacks.debugPrint != NULL)
{
va_list ap;
va_start(ap, pDebugString);
ADDR_DEBUGPRINT_INPUT debugPrintInput = {0};
debugPrintInput.size = sizeof(ADDR_DEBUGPRINT_INPUT);
debugPrintInput.pDebugString = const_cast<CHAR*>(pDebugString);
debugPrintInput.hClient = m_client.handle;
va_copy(debugPrintInput.ap, ap);
m_client.callbacks.debugPrint(&debugPrintInput);
va_end(ap);
va_end(debugPrintInput.ap);
}
#endif
}
} // Addr
} // namespace rocr
@@ -0,0 +1,79 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
****************************************************************************************************
* @file addrobject.h
* @brief Contains the Object base class definition.
****************************************************************************************************
*/
#ifndef __ADDR_OBJECT_H__
#define __ADDR_OBJECT_H__
#include "addrtypes.h"
#include "addrcommon.h"
namespace rocr {
namespace Addr
{
/**
****************************************************************************************************
* @brief This structure contains client specific data
****************************************************************************************************
*/
struct Client
{
ADDR_CLIENT_HANDLE handle;
ADDR_CALLBACKS callbacks;
};
/**
****************************************************************************************************
* @brief This class is the base class for all ADDR class objects.
****************************************************************************************************
*/
class Object
{
public:
Object();
Object(const Client* pClient);
virtual ~Object();
VOID* operator new(size_t size, VOID* pMem) noexcept;
VOID operator delete(VOID* pObj);
/// Microsoft compiler requires a matching delete implementation, which seems to be called when
/// bad_alloc is thrown. But currently C++ exception isn't allowed so a dummy implementation is
/// added to eliminate the warning.
VOID operator delete(VOID* pObj, VOID* pMem) { ADDR_ASSERT_ALWAYS(); }
VOID* Alloc(size_t size) const;
VOID Free(VOID* pObj) const;
VOID DebugPrint(const CHAR* pDebugString, ...) const;
const Client* GetClient() const {return &m_client;}
protected:
Client m_client;
static VOID* ClientAlloc(size_t size, const Client* pClient);
static VOID ClientFree(VOID* pObj, const Client* pClient);
private:
// disallow the copy constructor
Object(const Object& a);
// disallow the assignment operator
Object& operator=(const Object& a);
};
} // Addr
} // namespace rocr
#endif
@@ -0,0 +1,588 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
// Coordinate class implementation
#include "addrcommon.h"
#include "coord.h"
namespace rocr {
namespace Addr
{
namespace V2
{
Coordinate::Coordinate()
{
dim = DIM_X;
ord = 0;
}
Coordinate::Coordinate(enum Dim dim, INT_32 n)
{
set(dim, n);
}
VOID Coordinate::set(enum Dim d, INT_32 n)
{
dim = d;
ord = static_cast<INT_8>(n);
}
UINT_32 Coordinate::ison(const UINT_32 *coords) const
{
UINT_32 bit = static_cast<UINT_32>(1ull << static_cast<UINT_32>(ord));
return (coords[dim] & bit) ? 1 : 0;
}
enum Dim Coordinate::getdim()
{
return dim;
}
INT_8 Coordinate::getord()
{
return ord;
}
BOOL_32 Coordinate::operator==(const Coordinate& b)
{
return (dim == b.dim) && (ord == b.ord);
}
BOOL_32 Coordinate::operator<(const Coordinate& b)
{
BOOL_32 ret;
if (dim == b.dim)
{
ret = ord < b.ord;
}
else
{
if (dim == DIM_S || b.dim == DIM_M)
{
ret = TRUE;
}
else if (b.dim == DIM_S || dim == DIM_M)
{
ret = FALSE;
}
else if (ord == b.ord)
{
ret = dim < b.dim;
}
else
{
ret = ord < b.ord;
}
}
return ret;
}
BOOL_32 Coordinate::operator>(const Coordinate& b)
{
BOOL_32 lt = *this < b;
BOOL_32 eq = *this == b;
return !lt && !eq;
}
BOOL_32 Coordinate::operator<=(const Coordinate& b)
{
return (*this < b) || (*this == b);
}
BOOL_32 Coordinate::operator>=(const Coordinate& b)
{
return !(*this < b);
}
BOOL_32 Coordinate::operator!=(const Coordinate& b)
{
return !(*this == b);
}
Coordinate& Coordinate::operator++(INT_32)
{
ord++;
return *this;
}
// CoordTerm
CoordTerm::CoordTerm()
{
num_coords = 0;
}
VOID CoordTerm::Clear()
{
num_coords = 0;
}
VOID CoordTerm::add(Coordinate& co)
{
// This function adds a coordinate INT_32o the list
// It will prevent the same coordinate from appearing,
// and will keep the list ordered from smallest to largest
UINT_32 i;
for (i = 0; i < num_coords; i++)
{
if (m_coord[i] == co)
{
break;
}
if (m_coord[i] > co)
{
for (UINT_32 j = num_coords; j > i; j--)
{
m_coord[j] = m_coord[j - 1];
}
m_coord[i] = co;
num_coords++;
break;
}
}
if (i == num_coords)
{
m_coord[num_coords] = co;
num_coords++;
}
}
VOID CoordTerm::add(CoordTerm& cl)
{
for (UINT_32 i = 0; i < cl.num_coords; i++)
{
add(cl.m_coord[i]);
}
}
BOOL_32 CoordTerm::remove(Coordinate& co)
{
BOOL_32 remove = FALSE;
for (UINT_32 i = 0; i < num_coords; i++)
{
if (m_coord[i] == co)
{
remove = TRUE;
num_coords--;
}
if (remove)
{
m_coord[i] = m_coord[i + 1];
}
}
return remove;
}
BOOL_32 CoordTerm::Exists(Coordinate& co)
{
BOOL_32 exists = FALSE;
for (UINT_32 i = 0; i < num_coords; i++)
{
if (m_coord[i] == co)
{
exists = TRUE;
break;
}
}
return exists;
}
VOID CoordTerm::copyto(CoordTerm& cl)
{
cl.num_coords = num_coords;
for (UINT_32 i = 0; i < num_coords; i++)
{
cl.m_coord[i] = m_coord[i];
}
}
UINT_32 CoordTerm::getsize()
{
return num_coords;
}
UINT_32 CoordTerm::getxor(const UINT_32 *coords) const
{
UINT_32 out = 0;
for (UINT_32 i = 0; i < num_coords; i++)
{
out = out ^ m_coord[i].ison(coords);
}
return out;
}
VOID CoordTerm::getsmallest(Coordinate& co)
{
co = m_coord[0];
}
UINT_32 CoordTerm::Filter(INT_8 f, Coordinate& co, UINT_32 start, enum Dim axis)
{
for (UINT_32 i = start; i < num_coords;)
{
if (((f == '<' && m_coord[i] < co) ||
(f == '>' && m_coord[i] > co) ||
(f == '=' && m_coord[i] == co)) &&
(axis == NUM_DIMS || axis == m_coord[i].getdim()))
{
for (UINT_32 j = i; j < num_coords - 1; j++)
{
m_coord[j] = m_coord[j + 1];
}
num_coords--;
}
else
{
i++;
}
}
return num_coords;
}
Coordinate& CoordTerm::operator[](UINT_32 i)
{
return m_coord[i];
}
BOOL_32 CoordTerm::operator==(const CoordTerm& b)
{
BOOL_32 ret = TRUE;
if (num_coords != b.num_coords)
{
ret = FALSE;
}
else
{
for (UINT_32 i = 0; i < num_coords; i++)
{
// Note: the lists will always be in order, so we can compare the two lists at time
if (m_coord[i] != b.m_coord[i])
{
ret = FALSE;
break;
}
}
}
return ret;
}
BOOL_32 CoordTerm::operator!=(const CoordTerm& b)
{
return !(*this == b);
}
BOOL_32 CoordTerm::exceedRange(const UINT_32 *ranges)
{
BOOL_32 exceed = FALSE;
for (UINT_32 i = 0; (i < num_coords) && (exceed == FALSE); i++)
{
exceed = ((1u << m_coord[i].getord()) <= ranges[m_coord[i].getdim()]);
}
return exceed;
}
// coordeq
CoordEq::CoordEq()
{
m_numBits = 0;
}
VOID CoordEq::remove(Coordinate& co)
{
for (UINT_32 i = 0; i < m_numBits; i++)
{
m_eq[i].remove(co);
}
}
BOOL_32 CoordEq::Exists(Coordinate& co)
{
BOOL_32 exists = FALSE;
for (UINT_32 i = 0; i < m_numBits; i++)
{
if (m_eq[i].Exists(co))
{
exists = TRUE;
}
}
return exists;
}
VOID CoordEq::resize(UINT_32 n)
{
if (n > m_numBits)
{
for (UINT_32 i = m_numBits; i < n; i++)
{
m_eq[i].Clear();
}
}
m_numBits = n;
}
UINT_32 CoordEq::getsize()
{
return m_numBits;
}
UINT_64 CoordEq::solve(const UINT_32 *coords) const
{
UINT_64 out = 0;
for (UINT_32 i = 0; i < m_numBits; i++)
{
out |= static_cast<UINT_64>(m_eq[i].getxor(coords)) << i;
}
return out;
}
VOID CoordEq::solveAddr(
UINT_64 addr, UINT_32 sliceInM,
UINT_32 *coords) const
{
UINT_32 BitsValid[NUM_DIMS] = {0};
CoordEq temp = *this;
memset(coords, 0, NUM_DIMS * sizeof(coords[0]));
UINT_32 bitsLeft = 0;
for (UINT_32 i = 0; i < temp.m_numBits; i++)
{
UINT_32 termSize = temp.m_eq[i].getsize();
if (termSize == 1)
{
INT_8 bit = (addr >> i) & 1;
enum Dim dim = temp.m_eq[i][0].getdim();
INT_8 ord = temp.m_eq[i][0].getord();
ADDR_ASSERT((ord < 32) || (bit == 0));
BitsValid[dim] |= 1u << ord;
coords[dim] |= bit << ord;
temp.m_eq[i].Clear();
}
else if (termSize > 1)
{
bitsLeft++;
}
}
if (bitsLeft > 0)
{
if (sliceInM != 0)
{
coords[DIM_Z] = coords[DIM_M] / sliceInM;
BitsValid[DIM_Z] = 0xffffffff;
}
do
{
bitsLeft = 0;
for (UINT_32 i = 0; i < temp.m_numBits; i++)
{
UINT_32 termSize = temp.m_eq[i].getsize();
if (termSize == 1)
{
INT_8 bit = (addr >> i) & 1;
enum Dim dim = temp.m_eq[i][0].getdim();
INT_8 ord = temp.m_eq[i][0].getord();
ADDR_ASSERT((ord < 32) || (bit == 0));
ADDR_ASSERT(dim < DIM_S);
BitsValid[dim] |= 1u << ord;
coords[dim] |= bit << ord;
temp.m_eq[i].Clear();
}
else if (termSize > 1)
{
CoordTerm tmpTerm = temp.m_eq[i];
for (UINT_32 j = 0; j < termSize; j++)
{
enum Dim dim = temp.m_eq[i][j].getdim();
INT_8 ord = temp.m_eq[i][j].getord();
ADDR_ASSERT(dim < DIM_S);
if (BitsValid[dim] & (1u << ord))
{
UINT_32 v = (((coords[dim] >> ord) & 1) << i);
addr ^= static_cast<UINT_64>(v);
tmpTerm.remove(temp.m_eq[i][j]);
}
}
temp.m_eq[i] = tmpTerm;
bitsLeft++;
}
}
} while (bitsLeft > 0);
}
}
VOID CoordEq::copy(CoordEq& o, UINT_32 start, UINT_32 num)
{
o.m_numBits = (num == 0xFFFFFFFF) ? m_numBits : num;
for (UINT_32 i = 0; i < o.m_numBits; i++)
{
m_eq[start + i].copyto(o.m_eq[i]);
}
}
VOID CoordEq::reverse(UINT_32 start, UINT_32 num)
{
UINT_32 n = (num == 0xFFFFFFFF) ? m_numBits : num;
for (UINT_32 i = 0; i < n / 2; i++)
{
CoordTerm temp;
m_eq[start + i].copyto(temp);
m_eq[start + n - 1 - i].copyto(m_eq[start + i]);
temp.copyto(m_eq[start + n - 1 - i]);
}
}
VOID CoordEq::xorin(CoordEq& x, UINT_32 start)
{
UINT_32 n = ((m_numBits - start) < x.m_numBits) ? (m_numBits - start) : x.m_numBits;
for (UINT_32 i = 0; i < n; i++)
{
m_eq[start + i].add(x.m_eq[i]);
}
}
UINT_32 CoordEq::Filter(INT_8 f, Coordinate& co, UINT_32 start, enum Dim axis)
{
for (UINT_32 i = start; i < m_numBits;)
{
UINT_32 m = m_eq[i].Filter(f, co, 0, axis);
if (m == 0)
{
for (UINT_32 j = i; j < m_numBits - 1; j++)
{
m_eq[j] = m_eq[j + 1];
}
m_numBits--;
}
else
{
i++;
}
}
return m_numBits;
}
VOID CoordEq::shift(INT_32 amount, INT_32 start)
{
if (amount != 0)
{
INT_32 numBits = static_cast<INT_32>(m_numBits);
amount = -amount;
INT_32 inc = (amount < 0) ? -1 : 1;
INT_32 i = (amount < 0) ? numBits - 1 : start;
INT_32 end = (amount < 0) ? start - 1 : numBits;
for (; (inc > 0) ? i < end : i > end; i += inc)
{
if ((i + amount < start) || (i + amount >= numBits))
{
m_eq[i].Clear();
}
else
{
m_eq[i + amount].copyto(m_eq[i]);
}
}
}
}
CoordTerm& CoordEq::operator[](UINT_32 i)
{
return m_eq[i];
}
VOID CoordEq::mort2d(Coordinate& c0, Coordinate& c1, UINT_32 start, UINT_32 end)
{
if (end == 0)
{
ADDR_ASSERT(m_numBits > 0);
end = m_numBits - 1;
}
for (UINT_32 i = start; i <= end; i++)
{
UINT_32 select = (i - start) % 2;
Coordinate& c = (select == 0) ? c0 : c1;
m_eq[i].add(c);
c++;
}
}
VOID CoordEq::mort3d(Coordinate& c0, Coordinate& c1, Coordinate& c2, UINT_32 start, UINT_32 end)
{
if (end == 0)
{
ADDR_ASSERT(m_numBits > 0);
end = m_numBits - 1;
}
for (UINT_32 i = start; i <= end; i++)
{
UINT_32 select = (i - start) % 3;
Coordinate& c = (select == 0) ? c0 : ((select == 1) ? c1 : c2);
m_eq[i].add(c);
c++;
}
}
BOOL_32 CoordEq::operator==(const CoordEq& b)
{
BOOL_32 ret = TRUE;
if (m_numBits != b.m_numBits)
{
ret = FALSE;
}
else
{
for (UINT_32 i = 0; i < m_numBits; i++)
{
if (m_eq[i] != b.m_eq[i])
{
ret = FALSE;
break;
}
}
}
return ret;
}
BOOL_32 CoordEq::operator!=(const CoordEq& b)
{
return !(*this == b);
}
} // V2
} // Addr
} // namespace rocr
@@ -0,0 +1,130 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
// Class used to define a coordinate bit
#ifndef __COORD_H
#define __COORD_H
namespace rocr {
namespace Addr
{
namespace V2
{
#if defined(__cplusplus)
#if defined(_MSC_VER)
#if _MSC_VER >= 1900
#define ADDR_CPP11_COMPILER TRUE
#endif
#else
#if __cplusplus >= 201103L
#define ADDR_CPP11_COMPILER TRUE
#endif
#endif
#endif
#if defined(ADDR_CPP11_COMPILER)
enum Dim : INT_8
#else
enum Dim
#endif
{
DIM_X,
DIM_Y,
DIM_Z,
DIM_S,
DIM_M,
NUM_DIMS
};
class Coordinate
{
public:
Coordinate();
Coordinate(enum Dim dim, INT_32 n);
VOID set(enum Dim dim, INT_32 n);
UINT_32 ison(const UINT_32 *coords) const;
enum Dim getdim();
INT_8 getord();
BOOL_32 operator==(const Coordinate& b);
BOOL_32 operator<(const Coordinate& b);
BOOL_32 operator>(const Coordinate& b);
BOOL_32 operator<=(const Coordinate& b);
BOOL_32 operator>=(const Coordinate& b);
BOOL_32 operator!=(const Coordinate& b);
Coordinate& operator++(INT_32);
private:
enum Dim dim;
INT_8 ord;
};
class CoordTerm
{
public:
CoordTerm();
VOID Clear();
VOID add(Coordinate& co);
VOID add(CoordTerm& cl);
BOOL_32 remove(Coordinate& co);
BOOL_32 Exists(Coordinate& co);
VOID copyto(CoordTerm& cl);
UINT_32 getsize();
UINT_32 getxor(const UINT_32 *coords) const;
VOID getsmallest(Coordinate& co);
UINT_32 Filter(INT_8 f, Coordinate& co, UINT_32 start = 0, enum Dim axis = NUM_DIMS);
Coordinate& operator[](UINT_32 i);
BOOL_32 operator==(const CoordTerm& b);
BOOL_32 operator!=(const CoordTerm& b);
BOOL_32 exceedRange(const UINT_32 *ranges);
private:
static const UINT_32 MaxCoords = 8;
UINT_32 num_coords;
Coordinate m_coord[MaxCoords];
};
class CoordEq
{
public:
CoordEq();
VOID remove(Coordinate& co);
BOOL_32 Exists(Coordinate& co);
VOID resize(UINT_32 n);
UINT_32 getsize();
virtual UINT_64 solve(const UINT_32 *coords) const;
virtual VOID solveAddr(UINT_64 addr, UINT_32 sliceInM,
UINT_32 *coords) const;
VOID copy(CoordEq& o, UINT_32 start = 0, UINT_32 num = 0xFFFFFFFF);
VOID reverse(UINT_32 start = 0, UINT_32 num = 0xFFFFFFFF);
VOID xorin(CoordEq& x, UINT_32 start = 0);
UINT_32 Filter(INT_8 f, Coordinate& co, UINT_32 start = 0, enum Dim axis = NUM_DIMS);
VOID shift(INT_32 amount, INT_32 start = 0);
virtual CoordTerm& operator[](UINT_32 i);
VOID mort2d(Coordinate& c0, Coordinate& c1, UINT_32 start = 0, UINT_32 end = 0);
VOID mort3d(Coordinate& c0, Coordinate& c1, Coordinate& c2, UINT_32 start = 0, UINT_32 end = 0);
BOOL_32 operator==(const CoordEq& b);
BOOL_32 operator!=(const CoordEq& b);
private:
static const UINT_32 MaxEqBits = 64;
UINT_32 m_numBits;
CoordTerm m_eq[MaxEqBits];
};
} // V2
} // Addr
} // namespace rocr
#endif
@@ -0,0 +1,578 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
************************************************************************************************************************
* @file gfx10addrlib.h
* @brief Contains the Gfx10Lib class definition.
************************************************************************************************************************
*/
#ifndef __GFX10_ADDR_LIB_H__
#define __GFX10_ADDR_LIB_H__
#include "addrlib2.h"
#include "coord.h"
#include "gfx10SwizzlePattern.h"
namespace rocr {
namespace Addr
{
namespace V2
{
/**
************************************************************************************************************************
* @brief GFX10 specific settings structure.
************************************************************************************************************************
*/
struct Gfx10ChipSettings
{
struct
{
UINT_32 reserved1 : 32;
// Misc configuration bits
UINT_32 isDcn20 : 1; // If using DCN2.0
UINT_32 supportRbPlus : 1;
UINT_32 dsMipmapHtileFix : 1;
UINT_32 dccUnsup3DSwDis : 1;
UINT_32 : 4;
UINT_32 reserved2 : 24;
};
};
/**
************************************************************************************************************************
* @brief GFX10 data surface type.
************************************************************************************************************************
*/
enum Gfx10DataType
{
Gfx10DataColor,
Gfx10DataDepthStencil,
Gfx10DataFmask
};
const UINT_32 Gfx10LinearSwModeMask = (1u << ADDR_SW_LINEAR);
const UINT_32 Gfx10Blk256BSwModeMask = (1u << ADDR_SW_256B_S) |
(1u << ADDR_SW_256B_D);
const UINT_32 Gfx10Blk4KBSwModeMask = (1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_4KB_D_X);
const UINT_32 Gfx10Blk64KBSwModeMask = (1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx10BlkVarSwModeMask = (1u << ADDR_SW_VAR_Z_X) |
(1u << ADDR_SW_VAR_R_X);
const UINT_32 Gfx10ZSwModeMask = (1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_VAR_Z_X);
const UINT_32 Gfx10StandardSwModeMask = (1u << ADDR_SW_256B_S) |
(1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_64KB_S_X);
const UINT_32 Gfx10DisplaySwModeMask = (1u << ADDR_SW_256B_D) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_D_X);
const UINT_32 Gfx10RenderSwModeMask = (1u << ADDR_SW_64KB_R_X) |
(1u << ADDR_SW_VAR_R_X);
const UINT_32 Gfx10XSwModeMask = (1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X) |
Gfx10BlkVarSwModeMask;
const UINT_32 Gfx10TSwModeMask = (1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_D_T);
const UINT_32 Gfx10XorSwModeMask = Gfx10XSwModeMask |
Gfx10TSwModeMask;
const UINT_32 Gfx10Rsrc1dSwModeMask = Gfx10LinearSwModeMask |
Gfx10RenderSwModeMask |
Gfx10ZSwModeMask;
const UINT_32 Gfx10Rsrc2dSwModeMask = Gfx10LinearSwModeMask |
Gfx10Blk256BSwModeMask |
Gfx10Blk4KBSwModeMask |
Gfx10Blk64KBSwModeMask |
Gfx10BlkVarSwModeMask;
const UINT_32 Gfx10Rsrc3dSwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X) |
Gfx10BlkVarSwModeMask;
const UINT_32 Gfx10Rsrc2dPrtSwModeMask = (Gfx10Blk4KBSwModeMask | Gfx10Blk64KBSwModeMask) & ~Gfx10XSwModeMask;
const UINT_32 Gfx10Rsrc3dPrtSwModeMask = Gfx10Rsrc2dPrtSwModeMask & ~Gfx10DisplaySwModeMask;
const UINT_32 Gfx10Rsrc3dThin64KBSwModeMask = (1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx10Rsrc3dThinSwModeMask = Gfx10Rsrc3dThin64KBSwModeMask |
Gfx10BlkVarSwModeMask;
const UINT_32 Gfx10Rsrc3dViewAs2dSwModeMask = Gfx10Rsrc3dThinSwModeMask | Gfx10LinearSwModeMask;
const UINT_32 Gfx10Rsrc3dThickSwModeMask = Gfx10Rsrc3dSwModeMask & ~(Gfx10Rsrc3dThinSwModeMask | Gfx10LinearSwModeMask);
const UINT_32 Gfx10Rsrc3dThick4KBSwModeMask = Gfx10Rsrc3dThickSwModeMask & Gfx10Blk4KBSwModeMask;
const UINT_32 Gfx10Rsrc3dThick64KBSwModeMask = Gfx10Rsrc3dThickSwModeMask & Gfx10Blk64KBSwModeMask;
const UINT_32 Gfx10MsaaSwModeMask = (Gfx10ZSwModeMask |
Gfx10RenderSwModeMask)
;
const UINT_32 Dcn20NonBpp64SwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Dcn20Bpp64SwModeMask = (1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_D_X) |
Dcn20NonBpp64SwModeMask;
const UINT_32 Dcn21NonBpp64SwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Dcn21Bpp64SwModeMask = (1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_D_X) |
Dcn21NonBpp64SwModeMask;
/**
************************************************************************************************************************
* @brief This class is the GFX10 specific address library
* function set.
************************************************************************************************************************
*/
class Gfx10Lib : public Lib
{
public:
/// Creates Gfx10Lib object
static Addr::Lib* CreateObj(const Client* pClient)
{
VOID* pMem = Object::ClientAlloc(sizeof(Gfx10Lib), pClient);
return (pMem != NULL) ? new (pMem) Gfx10Lib(pClient) : NULL;
}
protected:
Gfx10Lib(const Client* pClient);
virtual ~Gfx10Lib();
virtual BOOL_32 HwlIsStandardSwizzle(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].isStd;
}
virtual BOOL_32 HwlIsDisplaySwizzle(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].isDisp;
}
virtual BOOL_32 HwlIsThin(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return ((IsTex1d(resourceType) == TRUE) ||
(IsTex2d(resourceType) == TRUE) ||
((IsTex3d(resourceType) == TRUE) &&
(m_swizzleModeTable[swizzleMode].isStd == FALSE) &&
(m_swizzleModeTable[swizzleMode].isDisp == FALSE)));
}
virtual BOOL_32 HwlIsThick(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return ((IsTex3d(resourceType) == TRUE) &&
(m_swizzleModeTable[swizzleMode].isStd || m_swizzleModeTable[swizzleMode].isDisp));
}
virtual ADDR_E_RETURNCODE HwlComputeHtileInfo(
const ADDR2_COMPUTE_HTILE_INFO_INPUT* pIn,
ADDR2_COMPUTE_HTILE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeCmaskInfo(
const ADDR2_COMPUTE_CMASK_INFO_INPUT* pIn,
ADDR2_COMPUTE_CMASK_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeDccInfo(
const ADDR2_COMPUTE_DCCINFO_INPUT* pIn,
ADDR2_COMPUTE_DCCINFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeCmaskAddrFromCoord(
const ADDR2_COMPUTE_CMASK_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_CMASK_ADDRFROMCOORD_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlComputeHtileAddrFromCoord(
const ADDR2_COMPUTE_HTILE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_HTILE_ADDRFROMCOORD_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlComputeHtileCoordFromAddr(
const ADDR2_COMPUTE_HTILE_COORDFROMADDR_INPUT* pIn,
ADDR2_COMPUTE_HTILE_COORDFROMADDR_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlSupportComputeDccAddrFromCoord(
const ADDR2_COMPUTE_DCC_ADDRFROMCOORD_INPUT* pIn);
virtual VOID HwlComputeDccAddrFromCoord(
const ADDR2_COMPUTE_DCC_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_DCC_ADDRFROMCOORD_OUTPUT* pOut);
virtual UINT_32 HwlGetEquationIndex(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual UINT_32 HwlGetEquationTableInfo(const ADDR_EQUATION** ppEquationTable) const
{
*ppEquationTable = m_equationTable;
return m_numEquations;
}
virtual ADDR_E_RETURNCODE HwlComputePipeBankXor(
const ADDR2_COMPUTE_PIPEBANKXOR_INPUT* pIn,
ADDR2_COMPUTE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSlicePipeBankXor(
const ADDR2_COMPUTE_SLICE_PIPEBANKXOR_INPUT* pIn,
ADDR2_COMPUTE_SLICE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSubResourceOffsetForSwizzlePattern(
const ADDR2_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_INPUT* pIn,
ADDR2_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeNonBlockCompressedView(
const ADDR2_COMPUTE_NONBLOCKCOMPRESSEDVIEW_INPUT* pIn,
ADDR2_COMPUTE_NONBLOCKCOMPRESSEDVIEW_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlGetPreferredSurfaceSetting(
const ADDR2_GET_PREFERRED_SURF_SETTING_INPUT* pIn,
ADDR2_GET_PREFERRED_SURF_SETTING_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoSanityCheck(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoLinear(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceAddrFromCoordTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
virtual UINT_32 HwlComputeMaxBaseAlignments() const;
virtual UINT_32 HwlComputeMaxMetaBaseAlignments() const;
virtual BOOL_32 HwlInitGlobalParams(const ADDR_CREATE_INPUT* pCreateIn);
virtual ChipFamily HwlConvertChipFamily(UINT_32 uChipFamily, UINT_32 uChipRevision);
private:
// Initialize equation table
VOID InitEquationTable();
ADDR_E_RETURNCODE ComputeSurfaceInfoMacroTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceInfoMicroTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoordMacroTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoordMicroTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
UINT_32 ComputeOffsetFromSwizzlePattern(
const UINT_64* pPattern,
UINT_32 numBits,
UINT_32 x,
UINT_32 y,
UINT_32 z,
UINT_32 s) const;
UINT_32 ComputeOffsetFromEquation(
const ADDR_EQUATION* pEq,
UINT_32 x,
UINT_32 y,
UINT_32 z) const;
ADDR_E_RETURNCODE ComputeStereoInfo(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
UINT_32* pAlignY,
UINT_32* pRightXor) const;
static void GetMipSize(
UINT_32 mip0Width,
UINT_32 mip0Height,
UINT_32 mip0Depth,
UINT_32 mipId,
UINT_32* pMipWidth,
UINT_32* pMipHeight,
UINT_32* pMipDepth = NULL)
{
*pMipWidth = ShiftCeil(Max(mip0Width, 1u), mipId);
*pMipHeight = ShiftCeil(Max(mip0Height, 1u), mipId);
if (pMipDepth != NULL)
{
*pMipDepth = ShiftCeil(Max(mip0Depth, 1u), mipId);
}
}
const ADDR_SW_PATINFO* GetSwizzlePatternInfo(
AddrSwizzleMode swizzleMode,
AddrResourceType resourceType,
UINT_32 log2Elem,
UINT_32 numFrag) const;
/**
* Will use the indices, "nibbles", to build an index equation inside pSwizzle
*
* @param pPatInfo Pointer to a patInfo. Contains indices mapping to the 2D nibble arrays which will be used to build an index equation.
* @param pSwizzle Array to write the index equation to.
*/
VOID GetSwizzlePatternFromPatternInfo(
const ADDR_SW_PATINFO* pPatInfo,
ADDR_BIT_SETTING (&pSwizzle)[20]) const
{
memcpy(pSwizzle,
GFX10_SW_PATTERN_NIBBLE01[pPatInfo->nibble01Idx],
sizeof(GFX10_SW_PATTERN_NIBBLE01[pPatInfo->nibble01Idx]));
memcpy(&pSwizzle[8],
GFX10_SW_PATTERN_NIBBLE2[pPatInfo->nibble2Idx],
sizeof(GFX10_SW_PATTERN_NIBBLE2[pPatInfo->nibble2Idx]));
memcpy(&pSwizzle[12],
GFX10_SW_PATTERN_NIBBLE3[pPatInfo->nibble3Idx],
sizeof(GFX10_SW_PATTERN_NIBBLE3[pPatInfo->nibble3Idx]));
memcpy(&pSwizzle[16],
GFX10_SW_PATTERN_NIBBLE4[pPatInfo->nibble4Idx],
sizeof(GFX10_SW_PATTERN_NIBBLE4[pPatInfo->nibble4Idx]));
}
VOID ConvertSwizzlePatternToEquation(
UINT_32 elemLog2,
AddrResourceType rsrcType,
AddrSwizzleMode swMode,
const ADDR_SW_PATINFO* pPatInfo,
ADDR_EQUATION* pEquation) const;
static INT_32 GetMetaElementSizeLog2(Gfx10DataType dataType);
static INT_32 GetMetaCacheSizeLog2(Gfx10DataType dataType);
void GetBlk256SizeLog2(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2,
Dim3d* pBlock) const;
void GetCompressedBlockSizeLog2(
Gfx10DataType dataType,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2,
Dim3d* pBlock) const;
INT_32 GetMetaOverlapLog2(
Gfx10DataType dataType,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2) const;
INT_32 Get3DMetaOverlapLog2(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2) const;
UINT_32 GetMetaBlkSize(
Gfx10DataType dataType,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2,
BOOL_32 pipeAlign,
Dim3d* pBlock) const;
INT_32 GetPipeRotateAmount(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const;
INT_32 GetEffectiveNumPipes() const
{
return ((m_settings.supportRbPlus == FALSE) ||
((m_numSaLog2 + 1) >= m_pipesLog2)) ? m_pipesLog2 : m_numSaLog2 + 1;
}
BOOL_32 IsRbAligned(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
const BOOL_32 isRtopt = IsRtOptSwizzle(swizzleMode);
const BOOL_32 isZ = IsZOrderSwizzle(swizzleMode);
const BOOL_32 isDisplay = IsDisplaySwizzle(swizzleMode);
return (IsTex2d(resourceType) && (isRtopt || isZ)) ||
(IsTex3d(resourceType) && isDisplay);
}
UINT_32 GetValidDisplaySwizzleModes(UINT_32 bpp) const;
BOOL_32 IsValidDisplaySwizzleMode(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
UINT_32 GetMaxNumMipsInTail(UINT_32 blockSizeLog2, BOOL_32 isThin) const;
static ADDR2_BLOCK_SET GetAllowedBlockSet(ADDR2_SWMODE_SET allowedSwModeSet, AddrResourceType rsrcType)
{
ADDR2_BLOCK_SET allowedBlockSet = {};
allowedBlockSet.micro = (allowedSwModeSet.value & Gfx10Blk256BSwModeMask) ? TRUE : FALSE;
allowedBlockSet.linear = (allowedSwModeSet.value & Gfx10LinearSwModeMask) ? TRUE : FALSE;
allowedBlockSet.var = (allowedSwModeSet.value & Gfx10BlkVarSwModeMask) ? TRUE : FALSE;
if (rsrcType == ADDR_RSRC_TEX_3D)
{
allowedBlockSet.macroThick4KB = (allowedSwModeSet.value & Gfx10Rsrc3dThick4KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThin64KB = (allowedSwModeSet.value & Gfx10Rsrc3dThin64KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThick64KB = (allowedSwModeSet.value & Gfx10Rsrc3dThick64KBSwModeMask) ? TRUE : FALSE;
}
else
{
allowedBlockSet.macroThin4KB = (allowedSwModeSet.value & Gfx10Blk4KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThin64KB = (allowedSwModeSet.value & Gfx10Blk64KBSwModeMask) ? TRUE : FALSE;
}
return allowedBlockSet;
}
static ADDR2_SWTYPE_SET GetAllowedSwSet(ADDR2_SWMODE_SET allowedSwModeSet)
{
ADDR2_SWTYPE_SET allowedSwSet = {};
allowedSwSet.sw_Z = (allowedSwModeSet.value & Gfx10ZSwModeMask) ? TRUE : FALSE;
allowedSwSet.sw_S = (allowedSwModeSet.value & Gfx10StandardSwModeMask) ? TRUE : FALSE;
allowedSwSet.sw_D = (allowedSwModeSet.value & Gfx10DisplaySwModeMask) ? TRUE : FALSE;
allowedSwSet.sw_R = (allowedSwModeSet.value & Gfx10RenderSwModeMask) ? TRUE : FALSE;
return allowedSwSet;
}
BOOL_32 IsInMipTail(
Dim3d mipTailDim,
UINT_32 maxNumMipsInTail,
UINT_32 mipWidth,
UINT_32 mipHeight,
UINT_32 numMipsToTheEnd) const
{
BOOL_32 inTail = ((mipWidth <= mipTailDim.w) &&
(mipHeight <= mipTailDim.h) &&
(numMipsToTheEnd <= maxNumMipsInTail));
return inTail;
}
UINT_32 GetBankXorBits(UINT_32 blockBits) const
{
return (blockBits > m_pipeInterleaveLog2 + m_pipesLog2 + ColumnBits) ?
Min(blockBits - m_pipeInterleaveLog2 - m_pipesLog2 - ColumnBits, BankBits) : 0;
}
BOOL_32 ValidateNonSwModeParams(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
BOOL_32 ValidateSwModeParams(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
static const UINT_32 ColumnBits = 2;
static const UINT_32 BankBits = 4;
static const UINT_32 UnalignedDccType = 3;
static const Dim3d Block256_3d[MaxNumOfBpp];
static const Dim3d Block64K_Log2_3d[MaxNumOfBpp];
static const Dim3d Block4K_Log2_3d[MaxNumOfBpp];
static const SwizzleModeFlags SwizzleModeTable[ADDR_SW_MAX_TYPE];
// Number of packers log2
UINT_32 m_numPkrLog2;
// Number of shader array log2
UINT_32 m_numSaLog2;
Gfx10ChipSettings m_settings;
UINT_32 m_colorBaseIndex;
UINT_32 m_xmaskBaseIndex;
UINT_32 m_htileBaseIndex;
UINT_32 m_dccBaseIndex;
};
} // V2
} // Addr
} // namespace rocr
#endif
@@ -0,0 +1,523 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
************************************************************************************************************************
* @file gfx11addrlib.h
* @brief Contains the Gfx11Lib class definition.
************************************************************************************************************************
*/
#ifndef __GFX11_ADDR_LIB_H__
#define __GFX11_ADDR_LIB_H__
#include "addrlib2.h"
#include "coord.h"
#include "gfx11SwizzlePattern.h"
namespace rocr {
namespace Addr
{
namespace V2
{
/**
************************************************************************************************************************
* @brief GFX11 specific settings structure.
************************************************************************************************************************
*/
struct Gfx11ChipSettings
{
struct
{
UINT_32 isGfx1150 : 1;
UINT_32 isGfx1103 : 1;
UINT_32 reserved1 : 30;
// Misc configuration bits
UINT_32 reserved2 : 32;
};
};
/**
************************************************************************************************************************
* @brief GFX11 data surface type.
************************************************************************************************************************
*/
enum Gfx11DataType
{
Gfx11DataColor,
Gfx11DataDepthStencil,
};
const UINT_32 Gfx11LinearSwModeMask = (1u << ADDR_SW_LINEAR);
const UINT_32 Gfx11Blk256BSwModeMask = (1u << ADDR_SW_256B_D);
const UINT_32 Gfx11Blk4KBSwModeMask = (1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_4KB_D_X);
const UINT_32 Gfx11Blk64KBSwModeMask = (1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx11Blk256KBSwModeMask = (1u << ADDR_SW_256KB_Z_X) |
(1u << ADDR_SW_256KB_S_X) |
(1u << ADDR_SW_256KB_D_X) |
(1u << ADDR_SW_256KB_R_X);
const UINT_32 Gfx11ZSwModeMask = (1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_256KB_Z_X);
const UINT_32 Gfx11StandardSwModeMask = (1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_256KB_S_X);
const UINT_32 Gfx11DisplaySwModeMask = (1u << ADDR_SW_256B_D) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_256KB_D_X);
const UINT_32 Gfx11RenderSwModeMask = (1u << ADDR_SW_64KB_R_X) |
(1u << ADDR_SW_256KB_R_X);
const UINT_32 Gfx11XSwModeMask = (1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X) |
Gfx11Blk256KBSwModeMask;
const UINT_32 Gfx11TSwModeMask = (1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_D_T);
const UINT_32 Gfx11XorSwModeMask = Gfx11XSwModeMask |
Gfx11TSwModeMask;
const UINT_32 Gfx11Rsrc1dSwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_64KB_R_X) |
(1u << ADDR_SW_64KB_Z_X) ;
const UINT_32 Gfx11Rsrc2dSwModeMask = Gfx11LinearSwModeMask |
Gfx11DisplaySwModeMask |
Gfx11ZSwModeMask |
Gfx11RenderSwModeMask;
const UINT_32 Gfx11Rsrc3dSwModeMask = Gfx11LinearSwModeMask |
Gfx11StandardSwModeMask |
Gfx11ZSwModeMask |
Gfx11RenderSwModeMask |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_256KB_D_X);
const UINT_32 Gfx11Rsrc2dPrtSwModeMask =
(Gfx11Blk4KBSwModeMask | Gfx11Blk64KBSwModeMask) & ~Gfx11XSwModeMask & Gfx11Rsrc2dSwModeMask;
const UINT_32 Gfx11Rsrc3dPrtSwModeMask =
(Gfx11Blk4KBSwModeMask | Gfx11Blk64KBSwModeMask) & ~Gfx11XSwModeMask & Gfx11Rsrc3dSwModeMask;
const UINT_32 Gfx11Rsrc3dThin64KBSwModeMask = (1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx11Rsrc3dThin256KBSwModeMask = (1u << ADDR_SW_256KB_Z_X) |
(1u << ADDR_SW_256KB_R_X);
const UINT_32 Gfx11Rsrc3dThinSwModeMask = Gfx11Rsrc3dThin64KBSwModeMask | Gfx11Rsrc3dThin256KBSwModeMask;
const UINT_32 Gfx11Rsrc3dThickSwModeMask = Gfx11Rsrc3dSwModeMask & ~(Gfx11Rsrc3dThinSwModeMask | Gfx11LinearSwModeMask);
const UINT_32 Gfx11Rsrc3dThick4KBSwModeMask = Gfx11Rsrc3dThickSwModeMask & Gfx11Blk4KBSwModeMask;
const UINT_32 Gfx11Rsrc3dThick64KBSwModeMask = Gfx11Rsrc3dThickSwModeMask & Gfx11Blk64KBSwModeMask;
const UINT_32 Gfx11Rsrc3dThick256KBSwModeMask = Gfx11Rsrc3dThickSwModeMask & Gfx11Blk256KBSwModeMask;
const UINT_32 Gfx11MsaaSwModeMask = Gfx11ZSwModeMask |
Gfx11RenderSwModeMask;
const UINT_32 Dcn32SwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X) |
(1u << ADDR_SW_256KB_D_X) |
(1u << ADDR_SW_256KB_R_X);
const UINT_32 Size256K = 262144u;
const UINT_32 Log2Size256K = 18u;
/**
************************************************************************************************************************
* @brief This class is the GFX11 specific address library
* function set.
************************************************************************************************************************
*/
class Gfx11Lib : public Lib
{
public:
/// Creates Gfx11Lib object
static Addr::Lib* CreateObj(const Client* pClient)
{
VOID* pMem = Object::ClientAlloc(sizeof(Gfx11Lib), pClient);
return (pMem != NULL) ? new (pMem) Gfx11Lib(pClient) : NULL;
}
protected:
Gfx11Lib(const Client* pClient);
virtual ~Gfx11Lib();
virtual BOOL_32 HwlIsStandardSwizzle(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].isStd;
}
virtual BOOL_32 HwlIsDisplaySwizzle(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].isDisp;
}
virtual BOOL_32 HwlIsThin(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return ((IsTex1d(resourceType) == TRUE) ||
(IsTex2d(resourceType) == TRUE) ||
((IsTex3d(resourceType) == TRUE) &&
(m_swizzleModeTable[swizzleMode].isStd == FALSE) &&
(m_swizzleModeTable[swizzleMode].isDisp == FALSE)));
}
virtual BOOL_32 HwlIsThick(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return ((IsTex3d(resourceType) == TRUE) &&
(m_swizzleModeTable[swizzleMode].isStd || m_swizzleModeTable[swizzleMode].isDisp));
}
virtual ADDR_E_RETURNCODE HwlComputeHtileInfo(
const ADDR2_COMPUTE_HTILE_INFO_INPUT* pIn,
ADDR2_COMPUTE_HTILE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeDccInfo(
const ADDR2_COMPUTE_DCCINFO_INPUT* pIn,
ADDR2_COMPUTE_DCCINFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeHtileAddrFromCoord(
const ADDR2_COMPUTE_HTILE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_HTILE_ADDRFROMCOORD_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlComputeHtileCoordFromAddr(
const ADDR2_COMPUTE_HTILE_COORDFROMADDR_INPUT* pIn,
ADDR2_COMPUTE_HTILE_COORDFROMADDR_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlSupportComputeDccAddrFromCoord(
const ADDR2_COMPUTE_DCC_ADDRFROMCOORD_INPUT* pIn);
virtual VOID HwlComputeDccAddrFromCoord(
const ADDR2_COMPUTE_DCC_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_DCC_ADDRFROMCOORD_OUTPUT* pOut);
virtual UINT_32 HwlGetEquationIndex(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual UINT_32 HwlGetEquationTableInfo(const ADDR_EQUATION** ppEquationTable) const
{
*ppEquationTable = m_equationTable;
return m_numEquations;
}
virtual ADDR_E_RETURNCODE HwlComputePipeBankXor(
const ADDR2_COMPUTE_PIPEBANKXOR_INPUT* pIn,
ADDR2_COMPUTE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSlicePipeBankXor(
const ADDR2_COMPUTE_SLICE_PIPEBANKXOR_INPUT* pIn,
ADDR2_COMPUTE_SLICE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSubResourceOffsetForSwizzlePattern(
const ADDR2_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_INPUT* pIn,
ADDR2_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeNonBlockCompressedView(
const ADDR2_COMPUTE_NONBLOCKCOMPRESSEDVIEW_INPUT* pIn,
ADDR2_COMPUTE_NONBLOCKCOMPRESSEDVIEW_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlGetPreferredSurfaceSetting(
const ADDR2_GET_PREFERRED_SURF_SETTING_INPUT* pIn,
ADDR2_GET_PREFERRED_SURF_SETTING_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlGetPossibleSwizzleModes(
const ADDR2_GET_PREFERRED_SURF_SETTING_INPUT* pIn,
ADDR2_GET_PREFERRED_SURF_SETTING_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlGetAllowedBlockSet(
ADDR2_SWMODE_SET allowedSwModeSet,
AddrResourceType rsrcType,
ADDR2_BLOCK_SET* pAllowedBlockSet) const;
virtual ADDR_E_RETURNCODE HwlGetAllowedSwSet(
ADDR2_SWMODE_SET allowedSwModeSet,
ADDR2_SWTYPE_SET* pAllowedSwSet) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoSanityCheck(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoLinear(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceAddrFromCoordTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
virtual UINT_32 HwlComputeMaxBaseAlignments() const;
virtual UINT_32 HwlComputeMaxMetaBaseAlignments() const;
virtual BOOL_32 HwlInitGlobalParams(const ADDR_CREATE_INPUT* pCreateIn);
virtual ChipFamily HwlConvertChipFamily(UINT_32 uChipFamily, UINT_32 uChipRevision);
private:
// Initialize equation table
VOID InitEquationTable();
ADDR_E_RETURNCODE ComputeSurfaceInfoMacroTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceInfoMicroTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoordMacroTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
ADDR_E_RETURNCODE ComputeSurfaceAddrFromCoordMicroTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
UINT_32 ComputeOffsetFromSwizzlePattern(
const UINT_64* pPattern,
UINT_32 numBits,
UINT_32 x,
UINT_32 y,
UINT_32 z,
UINT_32 s) const;
UINT_32 ComputeOffsetFromEquation(
const ADDR_EQUATION* pEq,
UINT_32 x,
UINT_32 y,
UINT_32 z) const;
ADDR_E_RETURNCODE ComputeStereoInfo(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
UINT_32* pAlignY,
UINT_32* pRightXor) const;
static void GetMipSize(
UINT_32 mip0Width,
UINT_32 mip0Height,
UINT_32 mip0Depth,
UINT_32 mipId,
UINT_32* pMipWidth,
UINT_32* pMipHeight,
UINT_32* pMipDepth = NULL)
{
*pMipWidth = ShiftCeil(Max(mip0Width, 1u), mipId);
*pMipHeight = ShiftCeil(Max(mip0Height, 1u), mipId);
if (pMipDepth != NULL)
{
*pMipDepth = ShiftCeil(Max(mip0Depth, 1u), mipId);
}
}
const ADDR_SW_PATINFO* GetSwizzlePatternInfo(
AddrSwizzleMode swizzleMode,
AddrResourceType resourceType,
UINT_32 log2Elem,
UINT_32 numFrag) const;
VOID GetSwizzlePatternFromPatternInfo(
const ADDR_SW_PATINFO* pPatInfo,
ADDR_BIT_SETTING (&pSwizzle)[20]) const
{
memcpy(pSwizzle,
GFX11_SW_PATTERN_NIBBLE01[pPatInfo->nibble01Idx],
sizeof(GFX11_SW_PATTERN_NIBBLE01[pPatInfo->nibble01Idx]));
memcpy(&pSwizzle[8],
GFX11_SW_PATTERN_NIBBLE2[pPatInfo->nibble2Idx],
sizeof(GFX11_SW_PATTERN_NIBBLE2[pPatInfo->nibble2Idx]));
memcpy(&pSwizzle[12],
GFX11_SW_PATTERN_NIBBLE3[pPatInfo->nibble3Idx],
sizeof(GFX11_SW_PATTERN_NIBBLE3[pPatInfo->nibble3Idx]));
memcpy(&pSwizzle[16],
GFX11_SW_PATTERN_NIBBLE4[pPatInfo->nibble4Idx],
sizeof(GFX11_SW_PATTERN_NIBBLE4[pPatInfo->nibble4Idx]));
}
VOID ConvertSwizzlePatternToEquation(
UINT_32 elemLog2,
AddrResourceType rsrcType,
AddrSwizzleMode swMode,
const ADDR_SW_PATINFO* pPatInfo,
ADDR_EQUATION* pEquation) const;
static INT_32 GetMetaElementSizeLog2(Gfx11DataType dataType);
static INT_32 GetMetaCacheSizeLog2(Gfx11DataType dataType);
void GetBlk256SizeLog2(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2,
Dim3d* pBlock) const;
void GetCompressedBlockSizeLog2(
Gfx11DataType dataType,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2,
Dim3d* pBlock) const;
INT_32 GetMetaOverlapLog2(
Gfx11DataType dataType,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2) const;
INT_32 Get3DMetaOverlapLog2(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2) const;
UINT_32 GetMetaBlkSize(
Gfx11DataType dataType,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 elemLog2,
UINT_32 numSamplesLog2,
BOOL_32 pipeAlign,
Dim3d* pBlock) const;
INT_32 GetPipeRotateAmount(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const;
INT_32 GetEffectiveNumPipes() const
{
return ((m_numSaLog2 + 1) >= m_pipesLog2) ? m_pipesLog2 : m_numSaLog2 + 1;
}
BOOL_32 IsRbAligned(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
const BOOL_32 isRtopt = IsRtOptSwizzle(swizzleMode);
const BOOL_32 isZ = IsZOrderSwizzle(swizzleMode);
const BOOL_32 isDisplay = IsDisplaySwizzle(swizzleMode);
return (IsTex2d(resourceType) && (isRtopt || isZ)) ||
(IsTex3d(resourceType) && isDisplay);
}
UINT_32 GetValidDisplaySwizzleModes(UINT_32 bpp) const;
BOOL_32 IsValidDisplaySwizzleMode(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
UINT_32 GetMaxNumMipsInTail(UINT_32 blockSizeLog2, BOOL_32 isThin) const;
BOOL_32 IsInMipTail(
Dim3d mipTailDim,
UINT_32 maxNumMipsInTail,
UINT_32 mipWidth,
UINT_32 mipHeight,
UINT_32 numMipsToTheEnd) const
{
BOOL_32 inTail = ((mipWidth <= mipTailDim.w) &&
(mipHeight <= mipTailDim.h) &&
(numMipsToTheEnd <= maxNumMipsInTail));
return inTail;
}
UINT_32 GetBankXorBits(UINT_32 blockBits) const
{
return (blockBits > m_pipeInterleaveLog2 + m_pipesLog2 + ColumnBits) ?
Min(blockBits - m_pipeInterleaveLog2 - m_pipesLog2 - ColumnBits, BankBits) : 0;
}
BOOL_32 ValidateNonSwModeParams(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
BOOL_32 ValidateSwModeParams(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
BOOL_32 IsBlock256kb(AddrSwizzleMode swizzleMode) const { return IsBlockVariable(swizzleMode); }
// TODO: figure out if there is any Column bits on GFX11...
static const UINT_32 ColumnBits = 2;
static const UINT_32 BankBits = 4;
static const UINT_32 UnalignedDccType = 3;
static const Dim3d Block256_3d[MaxNumOfBpp];
static const Dim3d Block256K_Log2_3d[MaxNumOfBpp];
static const Dim3d Block64K_Log2_3d[MaxNumOfBpp];
static const Dim3d Block4K_Log2_3d[MaxNumOfBpp];
static const SwizzleModeFlags SwizzleModeTable[ADDR_SW_MAX_TYPE];
// Number of packers log2
UINT_32 m_numPkrLog2;
// Number of shader array log2
UINT_32 m_numSaLog2;
Gfx11ChipSettings m_settings;
UINT_32 m_colorBaseIndex;
UINT_32 m_htileBaseIndex;
UINT_32 m_dccBaseIndex;
};
} // V2
} // Addr
} // namespace rocr
#endif
@@ -0,0 +1,280 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2023 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
************************************************************************************************************************
* @file gfx12SwizzlePattern.h
* @brief swizzle pattern for gfx12.
************************************************************************************************************************
*/
#ifndef __GFX12_SWIZZLE_PATTERN_H__
#define __GFX12_SWIZZLE_PATTERN_H__
namespace rocr {
namespace Addr
{
namespace V3
{
const ADDR_SW_PATINFO GFX12_SW_256B_2D_1xAA_PATINFO[] =
{
{ 0, 0, 0, 0, } , // 1 BPE @ SW_256B_2D_1xAA
{ 1, 0, 0, 0, } , // 2 BPE @ SW_256B_2D_1xAA
{ 2, 0, 0, 0, } , // 4 BPE @ SW_256B_2D_1xAA
{ 3, 0, 0, 0, } , // 8 BPE @ SW_256B_2D_1xAA
{ 4, 0, 0, 0, } , // 16 BPE @ SW_256B_2D_1xAA
};
const ADDR_SW_PATINFO GFX12_SW_256B_2D_2xAA_PATINFO[] =
{
{ 5, 0, 0, 0, } , // 1 BPE @ SW_256B_2D_2xAA
{ 6, 0, 0, 0, } , // 2 BPE @ SW_256B_2D_2xAA
{ 7, 0, 0, 0, } , // 4 BPE @ SW_256B_2D_2xAA
{ 8, 0, 0, 0, } , // 8 BPE @ SW_256B_2D_2xAA
{ 9, 0, 0, 0, } , // 16 BPE @ SW_256B_2D_2xAA
};
const ADDR_SW_PATINFO GFX12_SW_256B_2D_4xAA_PATINFO[] =
{
{ 10, 0, 0, 0, } , // 1 BPE @ SW_256B_2D_4xAA
{ 11, 0, 0, 0, } , // 2 BPE @ SW_256B_2D_4xAA
{ 12, 0, 0, 0, } , // 4 BPE @ SW_256B_2D_4xAA
{ 13, 0, 0, 0, } , // 8 BPE @ SW_256B_2D_4xAA
{ 14, 0, 0, 0, } , // 16 BPE @ SW_256B_2D_4xAA
};
const ADDR_SW_PATINFO GFX12_SW_256B_2D_8xAA_PATINFO[] =
{
{ 15, 0, 0, 0, } , // 1 BPE @ SW_256B_2D_8xAA
{ 16, 0, 0, 0, } , // 2 BPE @ SW_256B_2D_8xAA
{ 17, 0, 0, 0, } , // 4 BPE @ SW_256B_2D_8xAA
{ 18, 0, 0, 0, } , // 8 BPE @ SW_256B_2D_8xAA
{ 19, 0, 0, 0, } , // 16 BPE @ SW_256B_2D_8xAA
};
const ADDR_SW_PATINFO GFX12_SW_4KB_2D_1xAA_PATINFO[] =
{
{ 0, 1, 0, 0, } , // 1 BPE @ SW_4KB_2D_1xAA
{ 1, 2, 0, 0, } , // 2 BPE @ SW_4KB_2D_1xAA
{ 2, 3, 0, 0, } , // 4 BPE @ SW_4KB_2D_1xAA
{ 3, 4, 0, 0, } , // 8 BPE @ SW_4KB_2D_1xAA
{ 4, 5, 0, 0, } , // 16 BPE @ SW_4KB_2D_1xAA
};
const ADDR_SW_PATINFO GFX12_SW_4KB_2D_2xAA_PATINFO[] =
{
{ 5, 2, 0, 0, } , // 1 BPE @ SW_4KB_2D_2xAA
{ 6, 3, 0, 0, } , // 2 BPE @ SW_4KB_2D_2xAA
{ 7, 4, 0, 0, } , // 4 BPE @ SW_4KB_2D_2xAA
{ 8, 5, 0, 0, } , // 8 BPE @ SW_4KB_2D_2xAA
{ 9, 6, 0, 0, } , // 16 BPE @ SW_4KB_2D_2xAA
};
const ADDR_SW_PATINFO GFX12_SW_4KB_2D_4xAA_PATINFO[] =
{
{ 10, 3, 0, 0, } , // 1 BPE @ SW_4KB_2D_4xAA
{ 11, 4, 0, 0, } , // 2 BPE @ SW_4KB_2D_4xAA
{ 12, 5, 0, 0, } , // 4 BPE @ SW_4KB_2D_4xAA
{ 13, 6, 0, 0, } , // 8 BPE @ SW_4KB_2D_4xAA
{ 14, 7, 0, 0, } , // 16 BPE @ SW_4KB_2D_4xAA
};
const ADDR_SW_PATINFO GFX12_SW_4KB_2D_8xAA_PATINFO[] =
{
{ 15, 4, 0, 0, } , // 1 BPE @ SW_4KB_2D_8xAA
{ 16, 5, 0, 0, } , // 2 BPE @ SW_4KB_2D_8xAA
{ 17, 6, 0, 0, } , // 4 BPE @ SW_4KB_2D_8xAA
{ 18, 7, 0, 0, } , // 8 BPE @ SW_4KB_2D_8xAA
{ 19, 8, 0, 0, } , // 16 BPE @ SW_4KB_2D_8xAA
};
const ADDR_SW_PATINFO GFX12_SW_64KB_2D_1xAA_PATINFO[] =
{
{ 0, 1, 1, 0, } , // 1 BPE @ SW_64KB_2D_1xAA
{ 1, 2, 2, 0, } , // 2 BPE @ SW_64KB_2D_1xAA
{ 2, 3, 3, 0, } , // 4 BPE @ SW_64KB_2D_1xAA
{ 3, 4, 4, 0, } , // 8 BPE @ SW_64KB_2D_1xAA
{ 4, 5, 5, 0, } , // 16 BPE @ SW_64KB_2D_1xAA
};
const ADDR_SW_PATINFO GFX12_SW_64KB_2D_2xAA_PATINFO[] =
{
{ 5, 2, 2, 0, } , // 1 BPE @ SW_64KB_2D_2xAA
{ 6, 3, 3, 0, } , // 2 BPE @ SW_64KB_2D_2xAA
{ 7, 4, 4, 0, } , // 4 BPE @ SW_64KB_2D_2xAA
{ 8, 5, 5, 0, } , // 8 BPE @ SW_64KB_2D_2xAA
{ 9, 6, 6, 0, } , // 16 BPE @ SW_64KB_2D_2xAA
};
const ADDR_SW_PATINFO GFX12_SW_64KB_2D_4xAA_PATINFO[] =
{
{ 10, 3, 3, 0, } , // 1 BPE @ SW_64KB_2D_4xAA
{ 11, 4, 4, 0, } , // 2 BPE @ SW_64KB_2D_4xAA
{ 12, 5, 5, 0, } , // 4 BPE @ SW_64KB_2D_4xAA
{ 13, 6, 6, 0, } , // 8 BPE @ SW_64KB_2D_4xAA
{ 14, 7, 7, 0, } , // 16 BPE @ SW_64KB_2D_4xAA
};
const ADDR_SW_PATINFO GFX12_SW_64KB_2D_8xAA_PATINFO[] =
{
{ 15, 4, 4, 0, } , // 1 BPE @ SW_64KB_2D_8xAA
{ 16, 5, 5, 0, } , // 2 BPE @ SW_64KB_2D_8xAA
{ 17, 6, 6, 0, } , // 4 BPE @ SW_64KB_2D_8xAA
{ 18, 7, 7, 0, } , // 8 BPE @ SW_64KB_2D_8xAA
{ 19, 8, 8, 0, } , // 16 BPE @ SW_64KB_2D_8xAA
};
const ADDR_SW_PATINFO GFX12_SW_256KB_2D_1xAA_PATINFO[] =
{
{ 0, 1, 1, 1, } , // 1 BPE @ SW_256KB_2D_1xAA
{ 1, 2, 2, 2, } , // 2 BPE @ SW_256KB_2D_1xAA
{ 2, 3, 3, 3, } , // 4 BPE @ SW_256KB_2D_1xAA
{ 3, 4, 4, 4, } , // 8 BPE @ SW_256KB_2D_1xAA
{ 4, 5, 5, 5, } , // 16 BPE @ SW_256KB_2D_1xAA
};
const ADDR_SW_PATINFO GFX12_SW_256KB_2D_2xAA_PATINFO[] =
{
{ 5, 2, 2, 2, } , // 1 BPE @ SW_256KB_2D_2xAA
{ 6, 3, 3, 3, } , // 2 BPE @ SW_256KB_2D_2xAA
{ 7, 4, 4, 4, } , // 4 BPE @ SW_256KB_2D_2xAA
{ 8, 5, 5, 5, } , // 8 BPE @ SW_256KB_2D_2xAA
{ 9, 6, 6, 6, } , // 16 BPE @ SW_256KB_2D_2xAA
};
const ADDR_SW_PATINFO GFX12_SW_256KB_2D_4xAA_PATINFO[] =
{
{ 10, 3, 3, 3, } , // 1 BPE @ SW_256KB_2D_4xAA
{ 11, 4, 4, 4, } , // 2 BPE @ SW_256KB_2D_4xAA
{ 12, 5, 5, 5, } , // 4 BPE @ SW_256KB_2D_4xAA
{ 13, 6, 6, 6, } , // 8 BPE @ SW_256KB_2D_4xAA
{ 14, 7, 7, 7, } , // 16 BPE @ SW_256KB_2D_4xAA
};
const ADDR_SW_PATINFO GFX12_SW_256KB_2D_8xAA_PATINFO[] =
{
{ 15, 4, 4, 4, } , // 1 BPE @ SW_256KB_2D_8xAA
{ 16, 5, 5, 5, } , // 2 BPE @ SW_256KB_2D_8xAA
{ 17, 6, 6, 6, } , // 4 BPE @ SW_256KB_2D_8xAA
{ 18, 7, 7, 7, } , // 8 BPE @ SW_256KB_2D_8xAA
{ 19, 8, 8, 8, } , // 16 BPE @ SW_256KB_2D_8xAA
};
const ADDR_SW_PATINFO GFX12_SW_4KB_3D_PATINFO[] =
{
{ 20, 9, 0, 0, } , // 1 BPE @ SW_4KB_3D
{ 21, 10, 0, 0, } , // 2 BPE @ SW_4KB_3D
{ 22, 11, 0, 0, } , // 4 BPE @ SW_4KB_3D
{ 23, 12, 0, 0, } , // 8 BPE @ SW_4KB_3D
{ 24, 13, 0, 0, } , // 16 BPE @ SW_4KB_3D
};
const ADDR_SW_PATINFO GFX12_SW_64KB_3D_PATINFO[] =
{
{ 20, 9, 9, 0, } , // 1 BPE @ SW_64KB_3D
{ 21, 10, 10, 0, } , // 2 BPE @ SW_64KB_3D
{ 22, 11, 11, 0, } , // 4 BPE @ SW_64KB_3D
{ 23, 12, 12, 0, } , // 8 BPE @ SW_64KB_3D
{ 24, 13, 13, 0, } , // 16 BPE @ SW_64KB_3D
};
const ADDR_SW_PATINFO GFX12_SW_256KB_3D_PATINFO[] =
{
{ 20, 9, 9, 9, } , // 1 BPE @ SW_256KB_3D
{ 21, 10, 10, 9, } , // 2 BPE @ SW_256KB_3D
{ 22, 11, 11, 10, } , // 4 BPE @ SW_256KB_3D
{ 23, 12, 12, 11, } , // 8 BPE @ SW_256KB_3D
{ 24, 13, 13, 11, } , // 16 BPE @ SW_256KB_3D
};
const UINT_64 GFX12_SW_PATTERN_NIBBLE1[][8] =
{
{X0, X1, Y0, X2, Y1, Y2, X3, Y3, }, // 0
{0, X0, Y0, X1, Y1, X2, Y2, X3, }, // 1
{0, 0, X0, Y0, X1, Y1, X2, Y2, }, // 2
{0, 0, 0, X0, Y0, X1, X2, Y1, }, // 3
{0, 0, 0, 0, X0, Y0, X1, Y1, }, // 4
{S0, X0, Y0, X1, Y1, X2, Y2, X3, }, // 5
{0, S0, X0, Y0, X1, Y1, X2, Y2, }, // 6
{0, 0, S0, X0, Y0, X1, Y1, X2, }, // 7
{0, 0, 0, S0, X0, Y0, X1, Y1, }, // 8
{0, 0, 0, 0, S0, X0, Y0, X1, }, // 9
{S0, S1, X0, Y0, X1, Y1, X2, Y2, }, // 10
{0, S0, S1, X0, Y0, X1, Y1, X2, }, // 11
{0, 0, S0, S1, X0, Y0, X1, Y1, }, // 12
{0, 0, 0, S0, S1, X0, Y0, X1, }, // 13
{0, 0, 0, 0, S0, S1, X0, Y0, }, // 14
{S0, S1, S2, X0, Y0, X1, Y1, X2, }, // 15
{0, S0, S1, S2, X0, Y0, X1, Y1, }, // 16
{0, 0, S0, S1, S2, X0, Y0, X1, }, // 17
{0, 0, 0, S0, S1, S2, X0, Y0, }, // 18
{0, 0, 0, 0, S0, S1, S2, X0, }, // 19
{X0, X1, Z0, Y0, Y1, Z1, X2, Z2, }, // 20
{0, X0, Z0, Y0, X1, Z1, Y1, Z2, }, // 21
{0, 0, X0, Y0, X1, Z0, Y1, Z1, }, // 22
{0, 0, 0, X0, Y0, Z0, X1, Z1, }, // 23
{0, 0, 0, 0, X0, Z0, Y0, Z1, }, // 24
};
const UINT_64 GFX12_SW_PATTERN_NIBBLE2[][4] =
{
{0, 0, 0, 0, }, // 0
{Y4, X4, Y5, X5, }, // 1
{Y3, X4, Y4, X5, }, // 2
{Y3, X3, Y4, X4, }, // 3
{Y2, X3, Y3, X4, }, // 4
{Y2, X2, Y3, X3, }, // 5
{Y1, X2, Y2, X3, }, // 6
{Y1, X1, Y2, X2, }, // 7
{Y0, X1, Y1, X2, }, // 8
{Y2, X3, Z3, Y3, }, // 9
{Y2, X2, Z3, Y3, }, // 10
{Y2, X2, Z2, Y3, }, // 11
{Y1, X2, Z2, Y2, }, // 12
{Y1, X1, Z2, Y2, }, // 13
};
const UINT_64 GFX12_SW_PATTERN_NIBBLE3[][4] =
{
{0, 0, 0, 0, }, // 0
{Y6, X6, Y7, X7, }, // 1
{Y5, X6, Y6, X7, }, // 2
{Y5, X5, Y6, X6, }, // 3
{Y4, X5, Y5, X6, }, // 4
{Y4, X4, Y5, X5, }, // 5
{Y3, X4, Y4, X5, }, // 6
{Y3, X3, Y4, X4, }, // 7
{Y2, X3, Y3, X4, }, // 8
{X4, Z4, Y4, X5, }, // 9
{X3, Z4, Y4, X4, }, // 10
{X3, Z3, Y4, X4, }, // 11
{X3, Z3, Y3, X4, }, // 12
{X2, Z3, Y3, X3, }, // 13
};
const UINT_64 GFX12_SW_PATTERN_NIBBLE4[][2] =
{
{0, 0, }, // 0
{Y8, X8, }, // 1
{Y7, X8, }, // 2
{Y7, X7, }, // 3
{Y6, X7, }, // 4
{Y6, X6, }, // 5
{Y5, X6, }, // 6
{Y5, X5, }, // 7
{Y4, X5, }, // 8
{Z5, Y5, }, // 9
{Z4, Y5, }, // 10
{Z4, Y4, }, // 11
};
} // V3
} // Addr
} // namespace
#endif
@@ -0,0 +1,218 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2023 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
************************************************************************************************************************
* @file gfx12addrlib.h
* @brief Contains the Gfx12Lib class definition.
************************************************************************************************************************
*/
#ifndef __GFX12_ADDR_LIB_H__
#define __GFX12_ADDR_LIB_H__
#include "addrlib3.h"
#include "coord.h"
#include "gfx12SwizzlePattern.h"
namespace rocr {
namespace Addr
{
namespace V3
{
/**
************************************************************************************************************************
* @brief GFX12 specific settings structure.
************************************************************************************************************************
*/
struct Gfx12ChipSettings
{
struct
{
// Misc configuration bits
UINT_32 reserved : 32;
};
};
/**
************************************************************************************************************************
* @brief GFX12 data surface type.
************************************************************************************************************************
*/
/**
************************************************************************************************************************
* @brief This class is the GFX12 specific address library
* function set.
************************************************************************************************************************
*/
class Gfx12Lib : public Lib
{
public:
/// Creates Gfx12Lib object
static Addr::Lib* CreateObj(const Client* pClient)
{
VOID* pMem = Object::ClientAlloc(sizeof(Gfx12Lib), pClient);
return (pMem != NULL) ? new (pMem) Gfx12Lib(pClient) : NULL;
}
protected:
Gfx12Lib(const Client* pClient);
virtual ~Gfx12Lib();
// Meta surfaces such as Hi-S/Z are essentially images on GFX12, so just return the max
// image alignment.
virtual UINT_32 HwlComputeMaxMetaBaseAlignments() const { return 256 * 1024; }
UINT_32 GetMaxNumMipsInTail(
Addr3SwizzleMode swizzleMode,
UINT_32 blockSizeLog2) const;
BOOL_32 IsInMipTail(
const ADDR_EXTENT3D& mipTailDim,
const ADDR_EXTENT3D& mipDims,
UINT_32 maxNumMipsInTail,
UINT_32 numMipsToTheEnd) const
{
BOOL_32 inTail = ((mipDims.width <= mipTailDim.width) &&
(mipDims.height <= mipTailDim.height) &&
(numMipsToTheEnd <= maxNumMipsInTail));
return inTail;
}
virtual ADDR_E_RETURNCODE HwlComputeSurfaceAddrFromCoordTiled(
const ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeNonBlockCompressedView(
const ADDR3_COMPUTE_NONBLOCKCOMPRESSEDVIEW_INPUT* pIn,
ADDR3_COMPUTE_NONBLOCKCOMPRESSEDVIEW_OUTPUT* pOut) const;
virtual VOID HwlComputeSubResourceOffsetForSwizzlePattern(
const ADDR3_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_INPUT* pIn,
ADDR3_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSlicePipeBankXor(
const ADDR3_COMPUTE_SLICE_PIPEBANKXOR_INPUT* pIn,
ADDR3_COMPUTE_SLICE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual UINT_32 HwlGetEquationTableInfo(const ADDR_EQUATION** ppEquationTable) const
{
*ppEquationTable = m_equationTable;
return m_numEquations;
}
private:
Gfx12ChipSettings m_settings;
static const SwizzleModeFlags SwizzleModeTable[ADDR3_MAX_TYPE];
virtual ADDR_E_RETURNCODE HwlComputePipeBankXor(
const ADDR3_COMPUTE_PIPEBANKXOR_INPUT* pIn,
ADDR3_COMPUTE_PIPEBANKXOR_OUTPUT* pOut) const override;
virtual BOOL_32 HwlInitGlobalParams(const ADDR_CREATE_INPUT* pCreateIn) override;
void SanityCheckSurfSize(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
const ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
UINT_32 m_numSwizzleBits;
static const ADDR_EXTENT3D Block4K_Log2_3d[];
static const ADDR_EXTENT3D Block64K_Log2_3d[];
static const ADDR_EXTENT3D Block256K_Log2_3d[];
// Initialize equation table
VOID InitEquationTable();
VOID GetSwizzlePatternFromPatternInfo(
const ADDR_SW_PATINFO* pPatInfo,
ADDR_BIT_SETTING (&pSwizzle)[Log2Size256K]) const
{
memcpy(pSwizzle,
GFX12_SW_PATTERN_NIBBLE1[pPatInfo->nibble1Idx],
sizeof(GFX12_SW_PATTERN_NIBBLE1[pPatInfo->nibble1Idx]));
memcpy(&pSwizzle[8],
GFX12_SW_PATTERN_NIBBLE2[pPatInfo->nibble2Idx],
sizeof(GFX12_SW_PATTERN_NIBBLE2[pPatInfo->nibble2Idx]));
memcpy(&pSwizzle[12],
GFX12_SW_PATTERN_NIBBLE3[pPatInfo->nibble3Idx],
sizeof(GFX12_SW_PATTERN_NIBBLE3[pPatInfo->nibble3Idx]));
memcpy(&pSwizzle[16],
GFX12_SW_PATTERN_NIBBLE4[pPatInfo->nibble4Idx],
sizeof(GFX12_SW_PATTERN_NIBBLE4[pPatInfo->nibble4Idx]));
}
VOID ConvertSwizzlePatternToEquation(
UINT_32 elemLog2,
Addr3SwizzleMode swMode,
const ADDR_SW_PATINFO* pPatInfo,
ADDR_EQUATION* pEquation) const;
ADDR_EXTENT3D GetBaseMipExtents(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
ADDR_EXTENT3D GetBlockPixelDimensions(
Addr3SwizzleMode swizzleMode,
UINT_32 log2BytesPerPixel) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfo(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const override;
static ADDR_EXTENT3D GetMipExtent(
const ADDR_EXTENT3D& mip0,
UINT_32 mipId)
{
return {
ShiftCeil(Max(mip0.width, 1u), mipId),
ShiftCeil(Max(mip0.height, 1u), mipId),
ShiftCeil(Max(mip0.depth, 1u), mipId)
};
}
//# See 6.3 in //gfxip/gfx10/doc/architecture/ImageAddressing/gfx10_image_addressing.docx
// miptail is applied to only larger block size (4kb, 64kb, 256kb), so there is no miptail in linear and
// 256b_2d addressing since they are both 256b block.
BOOL_32 SupportsMipTail(Addr3SwizzleMode swizzleMode) const
{
return GetBlockSize(swizzleMode) > 256u;
}
UINT_32 ComputeOffsetFromEquation(
const ADDR_EQUATION* pEq,
UINT_32 x,
UINT_32 y,
UINT_32 z,
UINT_32 s) const;
const ADDR_SW_PATINFO* GetSwizzlePatternInfo(
Addr3SwizzleMode swizzleMode,
UINT_32 log2Elem,
UINT_32 numFrag) const;
VOID GetMipOffset(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
VOID GetMipOrigin(
const ADDR3_COMPUTE_SURFACE_INFO_INPUT* pIn,
const ADDR_EXTENT3D& mipExtentFirstInTail,
ADDR3_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
};
} // V3
} // Addr
} // namespace rocr
#endif
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,637 @@
/*
************************************************************************************************************************
*
* Copyright (C) 2007-2022 Advanced Micro Devices, Inc. All rights reserved.
* SPDX-License-Identifier: MIT
*
***********************************************************************************************************************/
/**
************************************************************************************************************************
* @file gfx9addrlib.h
* @brief Contgfx9ns the Gfx9Lib class definition.
************************************************************************************************************************
*/
#ifndef __GFX9_ADDR_LIB_H__
#define __GFX9_ADDR_LIB_H__
#include "addrlib2.h"
#include "coord.h"
namespace rocr {
namespace Addr
{
namespace V2
{
/**
************************************************************************************************************************
* @brief GFX9 specific settings structure.
************************************************************************************************************************
*/
struct Gfx9ChipSettings
{
struct
{
// Asic/Generation name
UINT_32 isArcticIsland : 1;
UINT_32 isVega10 : 1;
UINT_32 isRaven : 1;
UINT_32 isVega12 : 1;
UINT_32 isVega20 : 1;
UINT_32 reserved0 : 27;
// Display engine IP version name
UINT_32 isDce12 : 1;
UINT_32 isDcn1 : 1;
UINT_32 isDcn2 : 1;
UINT_32 reserved1 : 29;
// Misc configuration bits
UINT_32 metaBaseAlignFix : 1;
UINT_32 depthPipeXorDisable : 1;
UINT_32 htileAlignFix : 1;
UINT_32 applyAliasFix : 1;
UINT_32 htileCacheRbConflict: 1;
UINT_32 reserved2 : 27;
};
};
/**
************************************************************************************************************************
* @brief GFX9 data surface type.
************************************************************************************************************************
*/
enum Gfx9DataType
{
Gfx9DataColor,
Gfx9DataDepthStencil,
Gfx9DataFmask
};
const UINT_32 Gfx9LinearSwModeMask = (1u << ADDR_SW_LINEAR);
const UINT_32 Gfx9Blk256BSwModeMask = (1u << ADDR_SW_256B_S) |
(1u << ADDR_SW_256B_D) |
(1u << ADDR_SW_256B_R);
const UINT_32 Gfx9Blk4KBSwModeMask = (1u << ADDR_SW_4KB_Z) |
(1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_4KB_R) |
(1u << ADDR_SW_4KB_Z_X) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_4KB_R_X);
const UINT_32 Gfx9Blk64KBSwModeMask = (1u << ADDR_SW_64KB_Z) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_R) |
(1u << ADDR_SW_64KB_Z_T) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_R_T) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx9ZSwModeMask = (1u << ADDR_SW_4KB_Z) |
(1u << ADDR_SW_64KB_Z) |
(1u << ADDR_SW_64KB_Z_T) |
(1u << ADDR_SW_4KB_Z_X) |
(1u << ADDR_SW_64KB_Z_X);
const UINT_32 Gfx9StandardSwModeMask = (1u << ADDR_SW_256B_S) |
(1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_64KB_S_X);
const UINT_32 Gfx9DisplaySwModeMask = (1u << ADDR_SW_256B_D) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_D_X);
const UINT_32 Gfx9RotateSwModeMask = (1u << ADDR_SW_256B_R) |
(1u << ADDR_SW_4KB_R) |
(1u << ADDR_SW_64KB_R) |
(1u << ADDR_SW_64KB_R_T) |
(1u << ADDR_SW_4KB_R_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx9XSwModeMask = (1u << ADDR_SW_4KB_Z_X) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_4KB_R_X) |
(1u << ADDR_SW_64KB_Z_X) |
(1u << ADDR_SW_64KB_S_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Gfx9TSwModeMask = (1u << ADDR_SW_64KB_Z_T) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_R_T);
const UINT_32 Gfx9XorSwModeMask = Gfx9XSwModeMask |
Gfx9TSwModeMask;
const UINT_32 Gfx9AllSwModeMask = Gfx9LinearSwModeMask |
Gfx9ZSwModeMask |
Gfx9StandardSwModeMask |
Gfx9DisplaySwModeMask |
Gfx9RotateSwModeMask;
const UINT_32 Gfx9Rsrc1dSwModeMask = Gfx9LinearSwModeMask;
const UINT_32 Gfx9Rsrc2dSwModeMask = Gfx9AllSwModeMask;
const UINT_32 Gfx9Rsrc3dSwModeMask = Gfx9AllSwModeMask & ~Gfx9Blk256BSwModeMask & ~Gfx9RotateSwModeMask;
const UINT_32 Gfx9Rsrc2dPrtSwModeMask = (Gfx9Blk4KBSwModeMask | Gfx9Blk64KBSwModeMask) & ~Gfx9XSwModeMask;
const UINT_32 Gfx9Rsrc3dPrtSwModeMask = Gfx9Rsrc2dPrtSwModeMask & ~Gfx9RotateSwModeMask & ~Gfx9DisplaySwModeMask;
const UINT_32 Gfx9Rsrc3dThinSwModeMask = Gfx9DisplaySwModeMask & ~Gfx9Blk256BSwModeMask;
const UINT_32 Gfx9Rsrc3dThin4KBSwModeMask = Gfx9Rsrc3dThinSwModeMask & Gfx9Blk4KBSwModeMask;
const UINT_32 Gfx9Rsrc3dThin64KBSwModeMask = Gfx9Rsrc3dThinSwModeMask & Gfx9Blk64KBSwModeMask;
const UINT_32 Gfx9Rsrc3dThickSwModeMask = Gfx9Rsrc3dSwModeMask & ~(Gfx9Rsrc3dThinSwModeMask | Gfx9LinearSwModeMask);
const UINT_32 Gfx9Rsrc3dThick4KBSwModeMask = Gfx9Rsrc3dThickSwModeMask & Gfx9Blk4KBSwModeMask;
const UINT_32 Gfx9Rsrc3dThick64KBSwModeMask = Gfx9Rsrc3dThickSwModeMask & Gfx9Blk64KBSwModeMask;
const UINT_32 Gfx9MsaaSwModeMask = Gfx9AllSwModeMask & ~Gfx9Blk256BSwModeMask & ~Gfx9LinearSwModeMask;
const UINT_32 Dce12NonBpp32SwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_4KB_R) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_R) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_4KB_R_X) |
(1u << ADDR_SW_64KB_D_X) |
(1u << ADDR_SW_64KB_R_X);
const UINT_32 Dce12Bpp32SwModeMask = (1u << ADDR_SW_256B_D) |
(1u << ADDR_SW_256B_R) |
Dce12NonBpp32SwModeMask;
const UINT_32 Dcn1NonBpp64SwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_4KB_S) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_4KB_S_X) |
(1u << ADDR_SW_64KB_S_X);
const UINT_32 Dcn1Bpp64SwModeMask = (1u << ADDR_SW_4KB_D) |
(1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_4KB_D_X) |
(1u << ADDR_SW_64KB_D_X) |
Dcn1NonBpp64SwModeMask;
const UINT_32 Dcn2NonBpp64SwModeMask = (1u << ADDR_SW_LINEAR) |
(1u << ADDR_SW_64KB_S) |
(1u << ADDR_SW_64KB_S_T) |
(1u << ADDR_SW_64KB_S_X);
const UINT_32 Dcn2Bpp64SwModeMask = (1u << ADDR_SW_64KB_D) |
(1u << ADDR_SW_64KB_D_T) |
(1u << ADDR_SW_64KB_D_X) |
Dcn2NonBpp64SwModeMask;
/**
************************************************************************************************************************
* @brief GFX9 meta equation parameters
************************************************************************************************************************
*/
struct MetaEqParams
{
UINT_32 maxMip;
UINT_32 elementBytesLog2;
UINT_32 numSamplesLog2;
ADDR2_META_FLAGS metaFlag;
Gfx9DataType dataSurfaceType;
AddrSwizzleMode swizzleMode;
AddrResourceType resourceType;
UINT_32 metaBlkWidthLog2;
UINT_32 metaBlkHeightLog2;
UINT_32 metaBlkDepthLog2;
UINT_32 compBlkWidthLog2;
UINT_32 compBlkHeightLog2;
UINT_32 compBlkDepthLog2;
};
/**
************************************************************************************************************************
* @brief This class is the GFX9 specific address library
* function set.
************************************************************************************************************************
*/
class Gfx9Lib : public Lib
{
public:
/// Creates Gfx9Lib object
static Addr::Lib* CreateObj(const Client* pClient)
{
VOID* pMem = Object::ClientAlloc(sizeof(Gfx9Lib), pClient);
return (pMem != NULL) ? new (pMem) Gfx9Lib(pClient) : NULL;
}
protected:
Gfx9Lib(const Client* pClient);
virtual ~Gfx9Lib();
virtual BOOL_32 HwlIsStandardSwizzle(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return m_swizzleModeTable[swizzleMode].isStd ||
(IsTex3d(resourceType) && m_swizzleModeTable[swizzleMode].isDisp);
}
virtual BOOL_32 HwlIsDisplaySwizzle(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return IsTex2d(resourceType) && m_swizzleModeTable[swizzleMode].isDisp;
}
virtual BOOL_32 HwlIsThin(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return ((IsTex2d(resourceType) == TRUE) ||
((IsTex3d(resourceType) == TRUE) &&
(m_swizzleModeTable[swizzleMode].isZ == FALSE) &&
(m_swizzleModeTable[swizzleMode].isStd == FALSE)));
}
virtual BOOL_32 HwlIsThick(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const
{
return (IsTex3d(resourceType) &&
(m_swizzleModeTable[swizzleMode].isZ || m_swizzleModeTable[swizzleMode].isStd));
}
virtual ADDR_E_RETURNCODE HwlComputeHtileInfo(
const ADDR2_COMPUTE_HTILE_INFO_INPUT* pIn,
ADDR2_COMPUTE_HTILE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeCmaskInfo(
const ADDR2_COMPUTE_CMASK_INFO_INPUT* pIn,
ADDR2_COMPUTE_CMASK_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeDccInfo(
const ADDR2_COMPUTE_DCCINFO_INPUT* pIn,
ADDR2_COMPUTE_DCCINFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeCmaskAddrFromCoord(
const ADDR2_COMPUTE_CMASK_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_CMASK_ADDRFROMCOORD_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlComputeHtileAddrFromCoord(
const ADDR2_COMPUTE_HTILE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_HTILE_ADDRFROMCOORD_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlComputeHtileCoordFromAddr(
const ADDR2_COMPUTE_HTILE_COORDFROMADDR_INPUT* pIn,
ADDR2_COMPUTE_HTILE_COORDFROMADDR_OUTPUT* pOut);
virtual ADDR_E_RETURNCODE HwlSupportComputeDccAddrFromCoord(
const ADDR2_COMPUTE_DCC_ADDRFROMCOORD_INPUT* pIn);
virtual VOID HwlComputeDccAddrFromCoord(
const ADDR2_COMPUTE_DCC_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_DCC_ADDRFROMCOORD_OUTPUT* pOut);
virtual UINT_32 HwlGetEquationIndex(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeBlock256Equation(
AddrResourceType rsrcType,
AddrSwizzleMode swMode,
UINT_32 elementBytesLog2,
ADDR_EQUATION* pEquation) const;
virtual ADDR_E_RETURNCODE HwlComputeThinEquation(
AddrResourceType rsrcType,
AddrSwizzleMode swMode,
UINT_32 elementBytesLog2,
ADDR_EQUATION* pEquation) const;
virtual ADDR_E_RETURNCODE HwlComputeThickEquation(
AddrResourceType rsrcType,
AddrSwizzleMode swMode,
UINT_32 elementBytesLog2,
ADDR_EQUATION* pEquation) const;
// Get equation table pointer and number of equations
virtual UINT_32 HwlGetEquationTableInfo(const ADDR_EQUATION** ppEquationTable) const
{
*ppEquationTable = m_equationTable;
return m_numEquations;
}
virtual BOOL_32 IsEquationSupported(
AddrResourceType rsrcType,
AddrSwizzleMode swMode,
UINT_32 elementBytesLog2) const;
virtual ADDR_E_RETURNCODE HwlComputePipeBankXor(
const ADDR2_COMPUTE_PIPEBANKXOR_INPUT* pIn,
ADDR2_COMPUTE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSlicePipeBankXor(
const ADDR2_COMPUTE_SLICE_PIPEBANKXOR_INPUT* pIn,
ADDR2_COMPUTE_SLICE_PIPEBANKXOR_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSubResourceOffsetForSwizzlePattern(
const ADDR2_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_INPUT* pIn,
ADDR2_COMPUTE_SUBRESOURCE_OFFSET_FORSWIZZLEPATTERN_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlGetPreferredSurfaceSetting(
const ADDR2_GET_PREFERRED_SURF_SETTING_INPUT* pIn,
ADDR2_GET_PREFERRED_SURF_SETTING_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoSanityCheck(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoTiled(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceInfoLinear(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut) const;
virtual ADDR_E_RETURNCODE HwlComputeSurfaceAddrFromCoordTiled(
const ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_ADDRFROMCOORD_OUTPUT* pOut) const;
virtual UINT_32 HwlComputeMaxBaseAlignments() const;
virtual UINT_32 HwlComputeMaxMetaBaseAlignments() const;
virtual BOOL_32 HwlInitGlobalParams(const ADDR_CREATE_INPUT* pCreateIn);
virtual ChipFamily HwlConvertChipFamily(UINT_32 uChipFamily, UINT_32 uChipRevision);
virtual VOID ComputeThinBlockDimension(
UINT_32* pWidth,
UINT_32* pHeight,
UINT_32* pDepth,
UINT_32 bpp,
UINT_32 numSamples,
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode) const;
private:
VOID GetRbEquation(CoordEq* pRbEq, UINT_32 rbPerSeLog2, UINT_32 seLog2) const;
VOID GetDataEquation(CoordEq* pDataEq, Gfx9DataType dataSurfaceType,
AddrSwizzleMode swizzleMode, AddrResourceType resourceType,
UINT_32 elementBytesLog2, UINT_32 numSamplesLog2) const;
VOID GetPipeEquation(CoordEq* pPipeEq, CoordEq* pDataEq,
UINT_32 pipeInterleaveLog2, UINT_32 numPipesLog2,
UINT_32 numSamplesLog2, Gfx9DataType dataSurfaceType,
AddrSwizzleMode swizzleMode, AddrResourceType resourceType) const;
VOID GenMetaEquation(CoordEq* pMetaEq, UINT_32 maxMip,
UINT_32 elementBytesLog2, UINT_32 numSamplesLog2,
ADDR2_META_FLAGS metaFlag, Gfx9DataType dataSurfaceType,
AddrSwizzleMode swizzleMode, AddrResourceType resourceType,
UINT_32 metaBlkWidthLog2, UINT_32 metaBlkHeightLog2,
UINT_32 metaBlkDepthLog2, UINT_32 compBlkWidthLog2,
UINT_32 compBlkHeightLog2, UINT_32 compBlkDepthLog2) const;
const CoordEq* GetMetaEquation(const MetaEqParams& metaEqParams);
VOID GetMetaMipInfo(UINT_32 numMipLevels, Dim3d* pMetaBlkDim,
BOOL_32 dataThick, ADDR2_META_MIP_INFO* pInfo,
UINT_32 mip0Width, UINT_32 mip0Height, UINT_32 mip0Depth,
UINT_32* pNumMetaBlkX, UINT_32* pNumMetaBlkY, UINT_32* pNumMetaBlkZ) const;
BOOL_32 IsValidDisplaySwizzleMode(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
ADDR_E_RETURNCODE ComputeSurfaceLinearPadding(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
UINT_32* pMipmap0PaddedWidth,
UINT_32* pSlice0PaddedHeight,
ADDR2_MIP_INFO* pMipInfo = NULL) const;
static ADDR2_BLOCK_SET GetAllowedBlockSet(ADDR2_SWMODE_SET allowedSwModeSet, AddrResourceType rsrcType)
{
ADDR2_BLOCK_SET allowedBlockSet = {};
allowedBlockSet.micro = (allowedSwModeSet.value & Gfx9Blk256BSwModeMask) ? TRUE : FALSE;
allowedBlockSet.linear = (allowedSwModeSet.value & Gfx9LinearSwModeMask) ? TRUE : FALSE;
if (rsrcType == ADDR_RSRC_TEX_3D)
{
allowedBlockSet.macroThin4KB = (allowedSwModeSet.value & Gfx9Rsrc3dThin4KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThick4KB = (allowedSwModeSet.value & Gfx9Rsrc3dThick4KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThin64KB = (allowedSwModeSet.value & Gfx9Rsrc3dThin64KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThick64KB = (allowedSwModeSet.value & Gfx9Rsrc3dThick64KBSwModeMask) ? TRUE : FALSE;
}
else
{
allowedBlockSet.macroThin4KB = (allowedSwModeSet.value & Gfx9Blk4KBSwModeMask) ? TRUE : FALSE;
allowedBlockSet.macroThin64KB = (allowedSwModeSet.value & Gfx9Blk64KBSwModeMask) ? TRUE : FALSE;
}
return allowedBlockSet;
}
static ADDR2_SWTYPE_SET GetAllowedSwSet(ADDR2_SWMODE_SET allowedSwModeSet)
{
ADDR2_SWTYPE_SET allowedSwSet = {};
allowedSwSet.sw_Z = (allowedSwModeSet.value & Gfx9ZSwModeMask) ? TRUE : FALSE;
allowedSwSet.sw_S = (allowedSwModeSet.value & Gfx9StandardSwModeMask) ? TRUE : FALSE;
allowedSwSet.sw_D = (allowedSwModeSet.value & Gfx9DisplaySwModeMask) ? TRUE : FALSE;
allowedSwSet.sw_R = (allowedSwModeSet.value & Gfx9RotateSwModeMask) ? TRUE : FALSE;
return allowedSwSet;
}
BOOL_32 IsInMipTail(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
Dim3d mipTailDim,
UINT_32 width,
UINT_32 height,
UINT_32 depth) const
{
BOOL_32 inTail = ((width <= mipTailDim.w) &&
(height <= mipTailDim.h) &&
(IsThin(resourceType, swizzleMode) || (depth <= mipTailDim.d)));
return inTail;
}
BOOL_32 ValidateNonSwModeParams(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
BOOL_32 ValidateSwModeParams(const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn) const;
UINT_32 GetBankXorBits(UINT_32 macroBlockBits) const
{
UINT_32 pipeBits = GetPipeXorBits(macroBlockBits);
// Bank xor bits
UINT_32 bankBits = Min(macroBlockBits - pipeBits - m_pipeInterleaveLog2, m_banksLog2);
return bankBits;
}
UINT_32 ComputeSurfaceBaseAlignTiled(AddrSwizzleMode swizzleMode) const
{
UINT_32 baseAlign;
if (IsXor(swizzleMode))
{
baseAlign = GetBlockSize(swizzleMode);
}
else
{
baseAlign = 256;
}
return baseAlign;
}
// Initialize equation table
VOID InitEquationTable();
ADDR_E_RETURNCODE ComputeStereoInfo(
const ADDR2_COMPUTE_SURFACE_INFO_INPUT* pIn,
ADDR2_COMPUTE_SURFACE_INFO_OUTPUT* pOut,
UINT_32* pHeightAlign) const;
UINT_32 GetMipChainInfo(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 bpp,
UINT_32 mip0Width,
UINT_32 mip0Height,
UINT_32 mip0Depth,
UINT_32 blockWidth,
UINT_32 blockHeight,
UINT_32 blockDepth,
UINT_32 numMipLevel,
ADDR2_MIP_INFO* pMipInfo) const;
VOID GetMetaMiptailInfo(
ADDR2_META_MIP_INFO* pInfo,
Dim3d mipCoord,
UINT_32 numMipInTail,
Dim3d* pMetaBlkDim) const;
Dim3d GetMipStartPos(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 width,
UINT_32 height,
UINT_32 depth,
UINT_32 blockWidth,
UINT_32 blockHeight,
UINT_32 blockDepth,
UINT_32 mipId,
UINT_32 log2ElementBytes,
UINT_32* pMipTailBytesOffset) const;
AddrMajorMode GetMajorMode(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 mip0WidthInBlk,
UINT_32 mip0HeightInBlk,
UINT_32 mip0DepthInBlk) const
{
BOOL_32 yMajor = (mip0WidthInBlk < mip0HeightInBlk);
BOOL_32 xMajor = (yMajor == FALSE);
if (IsThick(resourceType, swizzleMode))
{
yMajor = yMajor && (mip0HeightInBlk >= mip0DepthInBlk);
xMajor = xMajor && (mip0WidthInBlk >= mip0DepthInBlk);
}
AddrMajorMode majorMode;
if (xMajor)
{
majorMode = ADDR_MAJOR_X;
}
else if (yMajor)
{
majorMode = ADDR_MAJOR_Y;
}
else
{
majorMode = ADDR_MAJOR_Z;
}
return majorMode;
}
Dim3d GetDccCompressBlk(
AddrResourceType resourceType,
AddrSwizzleMode swizzleMode,
UINT_32 bpp) const
{
UINT_32 index = Log2(bpp >> 3);
Dim3d compressBlkDim;
if (IsThin(resourceType, swizzleMode))
{
compressBlkDim.w = Block256_2d[index].w;
compressBlkDim.h = Block256_2d[index].h;
compressBlkDim.d = 1;
}
else if (IsStandardSwizzle(resourceType, swizzleMode))
{
compressBlkDim = Block256_3dS[index];
}
else
{
compressBlkDim = Block256_3dZ[index];
}
return compressBlkDim;
}
static const UINT_32 MaxSeLog2 = 3;
static const UINT_32 MaxRbPerSeLog2 = 2;
static const Dim3d Block256_3dS[MaxNumOfBpp];
static const Dim3d Block256_3dZ[MaxNumOfBpp];
static const UINT_32 MipTailOffset256B[];
static const SwizzleModeFlags SwizzleModeTable[ADDR_SW_MAX_TYPE];
static const UINT_32 MaxCachedMetaEq = 2;
Gfx9ChipSettings m_settings;
CoordEq m_cachedMetaEq[MaxCachedMetaEq];
MetaEqParams m_cachedMetaEqKey[MaxCachedMetaEq];
UINT_32 m_metaEqOverrideIndex;
};
} // V2
} // Addr
} // namespace rocr
#endif