Moved device code to mimic cuda header behavior
1. All fp32, fp64 math device/host functions should be in math_functions.h/.cpp 2. All fp32, fp64 fast math intrinsics for device/host functions should be in device_functions.h/.cpp 3. All the device code implementations should be in device_util.h/.cpp 4. Hence, made changes appropriately by moving code and creating new header files 5. Added math_functions.cpp/.h 6. Changed #ifndef signature to make sure no conflicts between headers with same names in hip/hip_runtime.h and hip/hcc_detail/hip_runtime.h 7. Changed tests to fit the code changes, making them to include appropriate headers 8. Added math_functions.cpp to CMakeLists.txt 9. Some of the tests are still broken, mostly host math functions will fix them in next commit 10. TODO: FIX compilation issues for host math functions Change-Id: I7a17637d7e294a7d224ffba932c1a08668febd26
This commit is contained in:
@@ -19,7 +19,8 @@ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
#include "hip/hip_runtime.h"
|
||||
#include <hip/hip_runtime.h>
|
||||
#include <hip/device_functions.h>
|
||||
#include "test_common.h"
|
||||
|
||||
#pragma GCC diagnostic ignored "-Wall"
|
||||
@@ -27,18 +28,18 @@ THE SOFTWARE.
|
||||
|
||||
__device__ void double_precision_intrinsics()
|
||||
{
|
||||
//__dadd_rd(0.0, 1.0);
|
||||
//__dadd_rn(0.0, 1.0);
|
||||
//__dadd_ru(0.0, 1.0);
|
||||
//__dadd_rz(0.0, 1.0);
|
||||
//__ddiv_rd(0.0, 1.0);
|
||||
//__ddiv_rn(0.0, 1.0);
|
||||
//__ddiv_ru(0.0, 1.0);
|
||||
//__ddiv_rz(0.0, 1.0);
|
||||
//__dmul_rd(1.0, 2.0);
|
||||
//__dmul_rn(1.0, 2.0);
|
||||
//__dmul_ru(1.0, 2.0);
|
||||
//__dmul_rz(1.0, 2.0);
|
||||
__dadd_rd(0.0, 1.0);
|
||||
__dadd_rn(0.0, 1.0);
|
||||
__dadd_ru(0.0, 1.0);
|
||||
__dadd_rz(0.0, 1.0);
|
||||
__ddiv_rd(0.0, 1.0);
|
||||
__ddiv_rn(0.0, 1.0);
|
||||
__ddiv_ru(0.0, 1.0);
|
||||
__ddiv_rz(0.0, 1.0);
|
||||
__dmul_rd(1.0, 2.0);
|
||||
__dmul_rn(1.0, 2.0);
|
||||
__dmul_ru(1.0, 2.0);
|
||||
__dmul_rz(1.0, 2.0);
|
||||
__drcp_rd(2.0);
|
||||
__drcp_rn(2.0);
|
||||
__drcp_ru(2.0);
|
||||
@@ -47,10 +48,10 @@ __device__ void double_precision_intrinsics()
|
||||
__dsqrt_rn(4.0);
|
||||
__dsqrt_ru(4.0);
|
||||
__dsqrt_rz(4.0);
|
||||
//__dsub_rd(2.0, 1.0);
|
||||
//__dsub_rn(2.0, 1.0);
|
||||
//__dsub_ru(2.0, 1.0);
|
||||
//__dsub_rz(2.0, 1.0);
|
||||
__dsub_rd(2.0, 1.0);
|
||||
__dsub_rn(2.0, 1.0);
|
||||
__dsub_ru(2.0, 1.0);
|
||||
__dsub_rz(2.0, 1.0);
|
||||
__fma_rd(1.0, 2.0, 3.0);
|
||||
__fma_rn(1.0, 2.0, 3.0);
|
||||
__fma_ru(1.0, 2.0, 3.0);
|
||||
|
||||
Reference in New Issue
Block a user