-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmatrix.h
More file actions
298 lines (212 loc) · 5.99 KB
/
Copy pathmatrix.h
File metadata and controls
298 lines (212 loc) · 5.99 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
#pragma once
#include <iostream>
#include <thread>
#include <condition_variable>
#include <mutex>
#include <vector>
#include <Windows.h>
#include "cuda.h"
#include "cuda_runtime.h"
#include "cuda_runtime_api.h"
#include "device_launch_parameters.h"
#define N_THREAD 1
#define DEBUG_M false
#define GPU true
#define CPU !GPU
#define N_STREAM 2
using namespace std;
typedef double dtype;
enum threadMode {SUM, SUM_ASSIGN, SCAL_SUM, SCAL_SUM_ASSIGN, SUB, SUB_ASSIGN, SCAL_SUB, SCAL_SUB_ASSIGN,
MULT, ELT_MULT, ELT_MULT_ASSIGN, SCAL_MULT, SCAL_MULT_ASSIGN, ELT_DIVIDE, ELT_DIVIDE_ASSIGN, SCAL_DIVIDE,
SCAL_DIVIDE_ASSIGN, SET, COPY, SCAL_POW, TRANS, ROW_SWAP, COL_SWAP, SIGMOID, NOT_EQUAL};
enum dimension { ROW, COL };
class Matrix {
private:
int mRow;
int mCol;
// For CPU
dtype* mData;
bool mHostAllocated;
bool mCpuIsChanged;
// For GPU
dtype* mDevData;
bool mDevAllocated;
bool mGpuIsChanged;
// For multithreading
static int numberOfMatrices;
static bool mInitThread;
static bool mThreadRun[N_THREAD];
static bool mThreadStart[N_THREAD];
static bool mThreadWait[N_THREAD];
static bool mThreadFinish[N_THREAD];
static mutex mThreadsReadyMtx[N_THREAD];
static mutex mNumberLock;
static int runThreads;
static condition_variable mThreadsReadyCV;
static vector<thread> mThreads;
static threadMode mMode;
static Matrix* threadM1;
static Matrix* threadM2;
static Matrix threadResult;
static Matrix* threadDestination;
static dtype threadScal;
static bool threadBool;
static bool* gpuBool;
static void* threadArg1;
static void* threadArg2;
static cudaStream_t cudaStream[N_STREAM];
static int currentStream;
// For gpu calculation
static bool gpuIsReady;
void readyForGpuCalc();
void initThreads();
void stopThreads();
static void threadFunction(int threadId);
void startGPUCalc();
void startThreads();
void calculation(threadMode aMode, Matrix* m1, Matrix* m2, Matrix* m_des = &threadResult, void* arg1 = NULL, void* arg2 = NULL);
void calculation(threadMode aMode, Matrix* m1, dtype scal, Matrix* m_des = &threadResult, void* arg1 = NULL, void* arg2 = NULL);
void allocHostData();
void freeHostData();
void freeDevData();
public:
Matrix() {
numberOfMatrices++;
mCpuIsChanged = false;
mGpuIsChanged = false;
mHostAllocated = false;
mDevAllocated = false;
if (CPU) initThreads();
else if (GPU) readyForGpuCalc();
mRow = 0;
mCol = 0;
mData = NULL;
}
Matrix(int aRow, int aCol) {
numberOfMatrices++;
mCpuIsChanged = false;
mGpuIsChanged = false;
mHostAllocated = false;
mDevAllocated = false;
if (CPU) initThreads();
else if (GPU) readyForGpuCalc();
mRow = aRow;
mCol = aCol;
allocHostData();
allocDevData();
}
Matrix(Matrix& other) {
numberOfMatrices++;
mCpuIsChanged = false;
mGpuIsChanged = false;
mHostAllocated = false;
mDevAllocated = false;
mRow = other.row();
mCol = other.col();
allocHostData();
allocDevData();
calculation(COPY, &other, nullptr, this);
}
~Matrix() {
numberOfMatrices--;
if(numberOfMatrices == 0 && CPU) stopThreads();
freeHostData();
freeDevData();
}
static void changeCudaStream() {
currentStream++;
if (currentStream == N_STREAM)
currentStream = 0;
}
__host__ __device__ int row() {
return mRow;
}
__host__ __device__ int col() {
return mCol;
}
__host__ __device__ int size() {
return mCol * mRow;
}
__host__ __device__ dtype* operator[](int i) {
if (i >= this->row()) {
cout << "Cannot access matrix[" << i << "]\n";
exit(EXIT_FAILURE);
}
retrieveDataFromDevice();
mCpuIsChanged = true;
return &mData[i * this->col()];
}
void operator=(Matrix& other);
Matrix matMult(Matrix* a, Matrix* b, Matrix* des = NULL, bool transA = false, bool transB = false);
Matrix matMult(dtype a, Matrix* b, Matrix* des = NULL);
Matrix matTranspose(Matrix* a);
Matrix matMean(Matrix* a, dimension d);
Matrix matCov(Matrix* a, dimension d);
Matrix matAdd(Matrix* a, Matrix* b);
Matrix matSub(Matrix* a, Matrix* b);
dtype matSum(Matrix* a);
dtype matDet(Matrix* a);
Matrix matInverse(Matrix* a);
Matrix matInverse2(Matrix* a);
Matrix matPseudoInverse(Matrix* a);
void rowSwap(int index1, int index2);
void colSwap(int index1, int index2);
void matAbs();
void operator()(int value);
void operator()(initializer_list<dtype> values, bool repeat = false);
Matrix operator+(Matrix& other);
Matrix operator-(Matrix& other);
void operator-=(Matrix& other);
void operator+=(Matrix& other);
void operator-=(dtype other);
void operator+=(dtype other);
// Element wise division
void operator/=(Matrix& other);
// Element wise division
void operator/=(dtype other);
// Element wise multiplication
void operator*=(Matrix& other);
// Element wise multiplication
void operator*=(dtype other);
dtype operator+(dtype other);
dtype operator-(dtype other);
bool operator==(Matrix& other);
bool operator!=(Matrix& other);
// Matrix multiplication
Matrix operator*(Matrix& other);
Matrix operator*(dtype other);
// Element wise division
Matrix operator/(Matrix& other);
// Element wise division
Matrix operator/(dtype other);
Matrix operator^(dtype coefficient);
Matrix cov(dimension d);
Matrix mean(dimension d);
// Transpose
Matrix T();
// Inverse
Matrix I();
// Pesudo Inverse
Matrix PI();
dtype det();
// Summation of matrix's all elements
dtype sum();
Matrix rowSum();
Matrix colSum();
Matrix repeat(int row, int col = 1);
// Data will be swiped out
void reshape(int aRow, int aCol);
void print();
void copyFrom(Matrix& source, int srcRowOffset = 0, int srcColOffset = 0);
void copyTo(Matrix& destination, int dstRowOffset = 0, int dstColOffset = 0);
// Split matrix, specific dimension with [aStart, aEnd)
Matrix split(dimension dimen, int aStart, int aSize);
// Concatenate this and other matrix
Matrix concat(dimension dimen, Matrix& other);
// Sigmoid
void sig();
bool saveToFile(string filePath);
bool loadFromFile(string filePath, bool forceReshape = false);
void allocDevData();
void retrieveDataFromDevice();
};