Repository navigation
Expand file tree
/
Copy pathkernel5.cu
More file actions
112 lines (100 loc) · 3.83 KB
/
Copy pathkernel5.cu
File metadata and controls
112 lines (100 loc) · 3.83 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
#include "common.h"
#define BLOCK_DIM 64
__global__ void spmspm_kernel5(COOMatrix *cooMatrix1,
CSRMatrix *csrMatrix1,
CSCMatrix *cscMatrix1,
COOMatrix *cooMatrix2,
CSRMatrix *csrMatrix2,
CSCMatrix *cscMatrix2,
COOMatrix *cooMatrix3,
const unsigned int numRows1,
const unsigned int numRows2,
const unsigned int numCols2,
const unsigned int numNonzeros1,
const unsigned int numNonzeros2)
{
extern __shared__ float row[];
__shared__ unsigned int nnz;
if (threadIdx.x == 0)
{
nnz = 0;
}
for (int i = threadIdx.x; i < numCols2; i += blockDim.x)
{
row[i] = 0;
}
__syncthreads();
unsigned int rowA = blockIdx.x;
unsigned int rowStart1 = csrMatrix1->rowPtrs[rowA];
unsigned int rowEnd1 = csrMatrix1->rowPtrs[rowA + 1];
for (unsigned int i = rowStart1 + threadIdx.x; i < rowEnd1; i += blockDim.x)
{
float valA = csrMatrix1->values[i];
unsigned int rowB = csrMatrix1->colIdxs[i];
unsigned int rowStart2 = csrMatrix2->rowPtrs[rowB];
unsigned int rowEnd2 = csrMatrix2->rowPtrs[rowB + 1];
for (unsigned int j = rowStart2; j < rowEnd2; ++j)
{
unsigned int colB = csrMatrix2->colIdxs[j];
float valB = csrMatrix2->values[j];
float val = valA * valB;
if (val != 0.0f)
{
float oldVal = atomicAdd(&row[colB], val);
if (oldVal == 0.0f)
{
atomicAdd(&nnz, 1);
}
}
}
}
__syncthreads();
if (nnz != 0)
{
__shared__ unsigned int idx;
if (threadIdx.x == 0)
{
idx = atomicAdd(&cooMatrix3->numNonzeros, nnz);
}
__syncthreads();
for (int i = threadIdx.x; i < numCols2; i += blockDim.x)
{
float v = row[i];
if (v != 0.0f)
{
unsigned int index = atomicAdd(&idx, 1);
cooMatrix3->rowIdxs[index] = rowA;
cooMatrix3->colIdxs[index] = i;
cooMatrix3->values[index] = v;
}
}
}
}
void spmspm_gpu5(COOMatrix *cooMatrix1,
CSRMatrix *csrMatrix1,
CSCMatrix *cscMatrix1,
COOMatrix *cooMatrix2,
CSRMatrix *csrMatrix2,
CSCMatrix *cscMatrix2,
COOMatrix *cooMatrix3,
unsigned int numRows1,
unsigned int numRows2,
unsigned int numCols2,
unsigned int numNonzeros1,
unsigned int numNonzeros2)
{
const dim3 block(BLOCK_DIM);
const dim3 grid(numRows1);
spmspm_kernel5<<<grid, block, numCols2 * sizeof(float)>>>(cooMatrix1,
csrMatrix1,
cscMatrix1,
cooMatrix2,
csrMatrix2,
cscMatrix2,
cooMatrix3,
numRows1,
numRows2,
numCols2,
numNonzeros1,
numNonzeros2);
}