Update docs 2025 04 14 (#54)
* Update docs 2025 03 31 - Docs: remove virtual_rocr.rst - Fix documentation warnings - Reformat HIP RTC - Docs: Refactor HIP porting guide - Docs: Expand HIP porting guide and CUDA driver porting guide - Minor fix - Docs: Update environment variables file - Bump rocm-docs-core[api_reference] from 1.15.0 to 1.17.0 in /docs/sphinx - Docs: Update FP8 page to show both FP8 and FP16 types - Bump sphinxcontrib-doxylink from 1.12.4 to 1.13.0 in /docs/sphinx - Bumps [rocm-docs-core[api_reference]](https://github.com/ROCm/rocm-docs-core) from 1.17.0 to 1.17.1. - Remove external link - Update programming model - Bump rocm-docs-core[api_reference] from 1.17.1 to 1.18.1 in /docs/sphinx - Docs: Add page for Complex Math API - Docs: Add page about HIP error codes - Update docs: the compilation cache is enabled by default - Fix fns32 function mask type in doc * Bump rocm-docs-core[api_reference] from 1.18.1 to 1.18.2 in /docs/sphinx Bumps [rocm-docs-core[api_reference]](https://github.com/ROCm/rocm-docs-core) from 1.18.1 to 1.18.2. - [Release notes](https://github.com/ROCm/rocm-docs-core/releases) - [Changelog](https://github.com/ROCm/rocm-docs-core/blob/develop/CHANGELOG.md) - [Commits](https://github.com/ROCm/rocm-docs-core/compare/v1.18.1...v1.18.2) --- updated-dependencies: - dependency-name: rocm-docs-core[api_reference] dependency-version: 1.18.2 dependency-type: direct:production update-type: version-update:semver-patch * Fix readme link * Docs: Fix verbose paths generated by doxygen * Handle git ssh in docs conf.py
This commit is contained in:
gecommit door
GitHub
bovenliggende
b301ef2282
commit
d0cf32a63a
@@ -69,34 +69,34 @@ better option, but is also limited in size.
|
||||
.. code-block:: cpp
|
||||
|
||||
__global__ void kernel_memory_allocation(TYPE* pointer){
|
||||
// The pointer is stored in shared memory, so that all
|
||||
// threads of the block can access the pointer
|
||||
__shared__ int *memory;
|
||||
// The pointer is stored in shared memory, so that all
|
||||
// threads of the block can access the pointer
|
||||
__shared__ int *memory;
|
||||
|
||||
size_t blockSize = blockDim.x;
|
||||
constexpr size_t elementsPerThread = 1024;
|
||||
if(threadIdx.x == 0){
|
||||
// allocate memory in one contiguous block
|
||||
memory = new int[blockDim.x * elementsPerThread];
|
||||
size_t blockSize = blockDim.x;
|
||||
constexpr size_t elementsPerThread = 1024;
|
||||
if(threadIdx.x == 0){
|
||||
// allocate memory in one contiguous block
|
||||
memory = new int[blockDim.x * elementsPerThread];
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// load pointer into thread-local variable to avoid
|
||||
// unnecessary accesses to shared memory
|
||||
int *localPtr = memory;
|
||||
|
||||
// work with allocated memory, e.g. initialization
|
||||
for(int i = 0; i < elementsPerThread; ++i){
|
||||
// access in a contiguous way
|
||||
localPtr[i * blockSize + threadIdx.x] = i;
|
||||
}
|
||||
|
||||
// synchronize to make sure no thread is accessing the memory before freeing
|
||||
__syncthreads();
|
||||
if(threadIdx.x == 0){
|
||||
delete[] memory;
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// load pointer into thread-local variable to avoid
|
||||
// unnecessary accesses to shared memory
|
||||
int *localPtr = memory;
|
||||
|
||||
// work with allocated memory, e.g. initialization
|
||||
for(int i = 0; i < elementsPerThread; ++i){
|
||||
// access in a contiguous way
|
||||
localPtr[i * blockSize + threadIdx.x] = i;
|
||||
}
|
||||
|
||||
// synchronize to make sure no thread is accessing the memory before freeing
|
||||
__syncthreads();
|
||||
if(threadIdx.x == 0){
|
||||
delete[] memory;
|
||||
}
|
||||
}
|
||||
|
||||
Copying between device and host
|
||||
--------------------------------------------------------------------------------
|
||||
|
||||
@@ -25,6 +25,10 @@ issue of reallocation when the extra buffer runs out.
|
||||
Virtual memory management solves these memory management problems. It helps to
|
||||
reduce memory usage and unnecessary ``memcpy`` calls.
|
||||
|
||||
HIP virtual memory management is built on top of HSA, which provides low-level
|
||||
access to AMD GPU memory. For more details on the underlying HSA runtime,
|
||||
see :doc:`ROCr documentation <rocr-runtime:index>`
|
||||
|
||||
.. _memory_allocation_virtual_memory:
|
||||
|
||||
Memory allocation
|
||||
|
||||
Verwijs in nieuw issue
Block a user