-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmain.cpp
More file actions
executable file
·124 lines (98 loc) · 3.1 KB
/
Copy pathmain.cpp
File metadata and controls
executable file
·124 lines (98 loc) · 3.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
#include "tensor_transpose.h"
#include "kernels.h"
#include <cutensor.h>
int main() {
int Nx = 512;
int Ny = 512;
int Nz = 512;
// Calculate grid size
int bSize = 8;
int nBx = Nx/bSize;
int nBy = Ny/bSize;
int nBz = Nz;
printf("n blocks : %d\n",nBx);
float *dIn_cpu;
float *dOut_cpu;
float *dIn_gpu;
float *dOut_gpu;
// Set up host data
dIn_cpu = (float*)malloc(Nx*Ny*Nz*sizeof(float));
dOut_cpu = (float*)malloc(Nx*Ny*Nz*sizeof(float));
// Initialize input
for( int k=0; k<Nz; k++ ){
for( int j=0; j<Ny; j++ ){
for( int i=0; i<Nx; i++ ){
dIn_cpu[T3D_INDEX(i,j,k,Nx,Ny,Nz)] = T3D_INDEX(i,j,k,Nx,Ny,Nz);
}
}
}
// Set up device data
hipMalloc(&dIn_gpu, Nx*Ny*Nz*sizeof(float));
hipMalloc(&dOut_gpu, Nx*Ny*Nz*sizeof(float));
// Copy dIn from host to device
hipMemcpy(dIn_gpu,dIn_cpu,Nx*Ny*Nz*sizeof(float),hipMemcpyHostToDevice);
// Call the transpose routine
// Loop to obtain timing results
for( int i=0; i<1000; i++ ){
transpose_021_fr<<<dim3(nBx,nBy,nBz),dim3(bSize,bSize,1)>>>(dIn_gpu,dOut_gpu,Nx,Ny,Nz);
}
for( int i=0; i<1000; i++ ){
transpose_102_fr<<<dim3(nBx,nBy,nBz),dim3(bSize,bSize,1)>>>(dIn_gpu,dOut_gpu,Nx,Ny,Nz);
}
for( int i=0; i<1000; i++ ){
transpose_120_fr<<<dim3(nBx,nBy,nBz),dim3(bSize,bSize,1)>>>(dIn_gpu,dOut_gpu,Nx,Ny,Nz);
}
for( int i=0; i<1000; i++ ){
transpose_201_fr<<<dim3(nBx,nBy,nBz),dim3(bSize,bSize,1)>>>(dIn_gpu,dOut_gpu,Nx,Ny,Nz);
}
for( int i=0; i<1000; i++ ){
transpose_210_fr<<<dim3(nBx,nBy,nBz),dim3(bSize,bSize,1)>>>(dIn_gpu,dOut_gpu,Nx,Ny,Nz);
}
for( int i=0; i<1000; i++ ){
copy_fr<<<dim3(nBx,nBy,nBz),dim3(bSize,bSize,1)>>>(dIn_gpu,dOut_gpu,Nx,Ny,Nz);
}
//
// Set up cuTensor for a 201 transpose
cutensorStatus_t err;
cutensorHandle_t handle;
cutensorInit(&handle);
cutensorTensorDescriptor_t descA,descB;
float one=1.0;
int64_t strideA[3],strideB[3];
int64_t extA[3],extB[3];
int imo1[] = {0,1,2};
int imc[] = {0,2,1};
extA[0] = Nx;
extA[1] = Ny;
extA[2] = Nz;
extB[0] = Nz;
extB[1] = Nx;
extB[2] = Ny;
strideA[0] = 1;
strideA[1] = Nx;
strideA[2] = Nx*Ny;
strideB[0] = 1;
strideB[1] = Nz;
strideB[2] = Nz*Nx;
err = cutensorInitTensorDescriptor( &handle, &descA,3,extA,strideA,CUDA_R_32F,CUTENSOR_OP_IDENTITY);
err = cutensorInitTensorDescriptor( &handle, &descB,3,extB,strideB,CUDA_R_32F,CUTENSOR_OP_IDENTITY);
if(err != CUTENSOR_STATUS_SUCCESS)
printf("Error while creating tensor descriptor\n");
for( int i=0; i<1000; i++ ){
err = cutensorPermutation( &handle, &one, dIn_gpu, &descA, imo1, dOut_gpu, &descB, imc,CUDA_R_32F, 0);
}
if(err != CUTENSOR_STATUS_SUCCESS)
printf("Error while permuting tensor: %s\n",cutensorGetErrorString(err));
const char *error = hipGetErrorString(hipGetLastError());
printf(error);
// Copy dOut from device to host
hipMemcpy(dOut_cpu,dOut_gpu,Nx*Ny*Nz*sizeof(float),hipMemcpyDeviceToHost);
hipDeviceSynchronize();
// Free device pointers
hipFree(dIn_gpu);
hipFree(dOut_gpu);
// Free host pointers
free(dIn_cpu);
free(dOut_cpu);
return 0;
}