CuPBoP/examples/dynamicSharedMemory/reverse.cu

// Get from: https://developer.nvidia.com/blog/using-shared-memory-cuda-cc/
#include <stdio.h>
#include <stdlib.h>

__global__ void dynamicReverse(int *d, int n)
{
  extern __shared__ int s[];
  int t = threadIdx.x;
  int tr = n-t-1;
  s[t] = d[t];
  __syncthreads();
  d[t] = s[tr];
}

int main()
{
  const int n = 64;
  int a[n], r[n], d[n];

  for (int i = 0; i < n; i++) {
    a[i] = i;
    r[i] = n-i-1;
    d[i] = 0;
  }

  int *d_d;
  cudaMalloc(&d_d, n * sizeof(int));

  // run version with static shared memory
  cudaMemcpy(d_d, a, n*sizeof(int), cudaMemcpyHostToDevice);
  dynamicReverse<<<1,n,n*sizeof(int)>>>(d_d, n);
  cudaMemcpy(d, d_d, n*sizeof(int), cudaMemcpyDeviceToHost);
  for (int i = 0; i < n; i++)
    if (d[i] != r[i]) {
      printf("Error: d[%d]!=r[%d] (%d, %d)n", i, i, d[i], r[i]);
      exit(1);
    }
    printf("PASS\n");
    return 0;
}
add static/dynamic shared memory example 2022-09-16 08:51:53 +08:00			`// Get from: https://developer.nvidia.com/blog/using-shared-memory-cuda-cc/`
			`#include <stdio.h>`
			`#include <stdlib.h>`

			`__global__ void dynamicReverse(int *d, int n)`
			`{`
			`extern __shared__ int s[];`
			`int t = threadIdx.x;`
			`int tr = n-t-1;`
			`s[t] = d[t];`
			`__syncthreads();`
			`d[t] = s[tr];`
			`}`

			`int main()`
			`{`
			`const int n = 64;`
			`int a[n], r[n], d[n];`

			`for (int i = 0; i < n; i++) {`
			`a[i] = i;`
			`r[i] = n-i-1;`
			`d[i] = 0;`
			`}`

			`int *d_d;`
			`cudaMalloc(&d_d, n * sizeof(int));`

			`// run version with static shared memory`
			`cudaMemcpy(d_d, a, n*sizeof(int), cudaMemcpyHostToDevice);`
			`dynamicReverse<<<1,n,n*sizeof(int)>>>(d_d, n);`
			`cudaMemcpy(d, d_d, n*sizeof(int), cudaMemcpyDeviceToHost);`
			`for (int i = 0; i < n; i++)`
			`if (d[i] != r[i]) {`
			`printf("Error: d[%d]!=r[%d] (%d, %d)n", i, i, d[i], r[i]);`
			`exit(1);`
			`}`
			`printf("PASS\n");`
			`return 0;`
			`}`