diff --git a/build/cuda.cu.o b/build/cuda.cu.o index 437afeb..21f4bd1 100644 Binary files a/build/cuda.cu.o and b/build/cuda.cu.o differ diff --git a/build/tensor.o b/build/tensor.o index 7b38f19..d3962d9 100644 Binary files a/build/tensor.o and b/build/tensor.o differ diff --git a/mprofile_20240605162109.dat b/mprofile_20240605162109.dat new file mode 100644 index 0000000..5e8562a --- /dev/null +++ b/mprofile_20240605162109.dat @@ -0,0 +1,28 @@ +CMDLINE /usr/bin/python3 train_singlegpu.py +MEM 1.109375 1717615269.6916 +MEM 14.167969 1717615269.7921 +MEM 21.085938 1717615269.8925 +MEM 26.617188 1717615269.9930 +MEM 32.312500 1717615270.0934 +MEM 38.429688 1717615270.1939 +MEM 40.683594 1717615270.2943 +MEM 46.445312 1717615270.3947 +MEM 54.089844 1717615270.4953 +MEM 58.347656 1717615270.5959 +MEM 63.308594 1717615270.6964 +MEM 65.226562 1717615270.7970 +MEM 70.898438 1717615270.8975 +MEM 76.828125 1717615270.9980 +MEM 82.242188 1717615271.0986 +MEM 87.656250 1717615271.1991 +MEM 93.328125 1717615271.2996 +MEM 98.742188 1717615271.4001 +MEM 104.671875 1717615271.5006 +MEM 135.867188 1717615271.6011 +MEM 117.457031 1717615271.7016 +MEM 125.246094 1717615271.8022 +MEM 125.503906 1717615271.9026 +MEM 125.503906 1717615272.0029 +MEM 127.074219 1717615272.1032 +MEM 127.074219 1717615272.2036 +MEM 73.593750 1717615272.3040 diff --git a/mprofile_20240605162319.dat b/mprofile_20240605162319.dat new file mode 100644 index 0000000..b859d5e --- /dev/null +++ b/mprofile_20240605162319.dat @@ -0,0 +1,257 @@ +CMDLINE /usr/bin/python3 train_singlegpu.py +MEM 1.171875 1717615399.9716 +MEM 18.777344 1717615400.0720 +MEM 25.433594 1717615400.1725 +MEM 30.570312 1717615400.2729 +MEM 36.207031 1717615400.3734 +MEM 40.167969 1717615400.4738 +MEM 46.121094 1717615400.5743 +MEM 56.503906 1717615400.6747 +MEM 63.429688 1717615400.7751 +MEM 70.007812 1717615400.8755 +MEM 84.187500 1717615400.9758 +MEM 98.109375 1717615401.0761 +MEM 140.132812 1717615401.1765 +MEM 125.375000 1717615401.2769 +MEM 125.632812 1717615401.3772 +MEM 126.433594 1717615401.4775 +MEM 127.203125 1717615401.5778 +MEM 127.203125 1717615401.6781 +MEM 127.316406 1717615401.7785 +MEM 127.316406 1717615401.8787 +MEM 134.515625 1717615401.9790 +MEM 134.773438 1717615402.0793 +MEM 136.054688 1717615402.1796 +MEM 140.953125 1717615402.2799 +MEM 140.953125 1717615402.3802 +MEM 146.867188 1717615402.4805 +MEM 147.125000 1717615402.5809 +MEM 153.562500 1717615402.6813 +MEM 154.335938 1717615402.7817 +MEM 154.335938 1717615402.8820 +MEM 154.335938 1717615402.9825 +MEM 154.335938 1717615403.0828 +MEM 154.335938 1717615403.1832 +MEM 154.335938 1717615403.2836 +MEM 161.039062 1717615403.3839 +MEM 161.812500 1717615403.4842 +MEM 162.328125 1717615403.5845 +MEM 169.273438 1717615403.6848 +MEM 169.531250 1717615403.7852 +MEM 176.992188 1717615403.8855 +MEM 176.992188 1717615403.9857 +MEM 184.449219 1717615404.0860 +MEM 184.449219 1717615404.1863 +MEM 186.488281 1717615404.2866 +MEM 191.898438 1717615404.3869 +MEM 192.156250 1717615404.4872 +MEM 199.632812 1717615404.5875 +MEM 199.632812 1717615404.6878 +MEM 207.109375 1717615404.7881 +MEM 207.109375 1717615404.8884 +MEM 213.296875 1717615404.9887 +MEM 214.761719 1717615405.0890 +MEM 214.957031 1717615405.1894 +MEM 222.171875 1717615405.2897 +MEM 222.429688 1717615405.3900 +MEM 229.648438 1717615405.4903 +MEM 229.906250 1717615405.5906 +MEM 236.093750 1717615405.6910 +MEM 237.382812 1717615405.7913 +MEM 237.898438 1717615405.8916 +MEM 244.859375 1717615405.9919 +MEM 245.117188 1717615406.0922 +MEM 252.335938 1717615406.1925 +MEM 252.593750 1717615406.2928 +MEM 258.781250 1717615406.3931 +MEM 260.070312 1717615406.4934 +MEM 260.585938 1717615406.5937 +MEM 267.546875 1717615406.6940 +MEM 267.804688 1717615406.7944 +MEM 275.023438 1717615406.8947 +MEM 275.281250 1717615406.9949 +MEM 281.726562 1717615407.0953 +MEM 282.757812 1717615407.1956 +MEM 283.273438 1717615407.2959 +MEM 290.234375 1717615407.3962 +MEM 290.234375 1717615407.4965 +MEM 297.710938 1717615407.5969 +MEM 297.710938 1717615407.6972 +MEM 298.226562 1717615407.7976 +MEM 305.187500 1717615407.8979 +MEM 305.445312 1717615407.9983 +MEM 305.445312 1717615408.0987 +MEM 312.921875 1717615408.1990 +MEM 312.921875 1717615408.2993 +MEM 319.109375 1717615408.3996 +MEM 320.398438 1717615408.4999 +MEM 320.656250 1717615408.6002 +MEM 327.875000 1717615408.7005 +MEM 328.132812 1717615408.8008 +MEM 335.609375 1717615408.9011 +MEM 335.609375 1717615409.0014 +MEM 341.796875 1717615409.1018 +MEM 343.085938 1717615409.2021 +MEM 343.601562 1717615409.3024 +MEM 350.562500 1717615409.4026 +MEM 350.820312 1717615409.5029 +MEM 358.039062 1717615409.6032 +MEM 358.296875 1717615409.7035 +MEM 365.000000 1717615409.8039 +MEM 365.773438 1717615409.9042 +MEM 366.289062 1717615410.0044 +MEM 373.250000 1717615410.1047 +MEM 373.507812 1717615410.2051 +MEM 380.726562 1717615410.3055 +MEM 380.984375 1717615410.4057 +MEM 388.460938 1717615410.5060 +MEM 388.460938 1717615410.6064 +MEM 390.523438 1717615410.7066 +MEM 395.937500 1717615410.8069 +MEM 396.195312 1717615410.9072 +MEM 403.414062 1717615411.0075 +MEM 403.671875 1717615411.1078 +MEM 411.148438 1717615411.2081 +MEM 411.148438 1717615411.3084 +MEM 417.335938 1717615411.4088 +MEM 418.625000 1717615411.5091 +MEM 418.882812 1717615411.6094 +MEM 426.101562 1717615411.7097 +MEM 426.359375 1717615411.8100 +MEM 433.578125 1717615411.9103 +MEM 433.835938 1717615412.0106 +MEM 440.023438 1717615412.1110 +MEM 441.312500 1717615412.2113 +MEM 441.828125 1717615412.3116 +MEM 448.789062 1717615412.4119 +MEM 449.046875 1717615412.5122 +MEM 456.265625 1717615412.6125 +MEM 456.523438 1717615412.7127 +MEM 462.968750 1717615412.8131 +MEM 464.000000 1717615412.9133 +MEM 464.515625 1717615413.0136 +MEM 471.476562 1717615413.1139 +MEM 471.734375 1717615413.2142 +MEM 478.949219 1717615413.3145 +MEM 479.207031 1717615413.4148 +MEM 486.683594 1717615413.5151 +MEM 486.683594 1717615413.6154 +MEM 487.457031 1717615413.7157 +MEM 494.160156 1717615413.8160 +MEM 494.417969 1717615413.9163 +MEM 501.636719 1717615414.0166 +MEM 501.894531 1717615414.1169 +MEM 509.113281 1717615414.2172 +MEM 509.371094 1717615414.3175 +MEM 509.886719 1717615414.4178 +MEM 516.847656 1717615414.5181 +MEM 516.847656 1717615414.6184 +MEM 524.324219 1717615414.7187 +MEM 524.582031 1717615414.8190 +MEM 531.027344 1717615414.9193 +MEM 532.058594 1717615415.0196 +MEM 532.574219 1717615415.1199 +MEM 539.535156 1717615415.2202 +MEM 539.535156 1717615415.3205 +MEM 547.011719 1717615415.4209 +MEM 547.269531 1717615415.5211 +MEM 554.488281 1717615415.6214 +MEM 554.746094 1717615415.7217 +MEM 556.808594 1717615415.8220 +MEM 562.222656 1717615415.9223 +MEM 562.222656 1717615416.0226 +MEM 569.699219 1717615416.1229 +MEM 569.957031 1717615416.2232 +MEM 577.175781 1717615416.3235 +MEM 577.433594 1717615416.4238 +MEM 582.332031 1717615416.5241 +MEM 584.910156 1717615416.6244 +MEM 584.910156 1717615416.7248 +MEM 592.386719 1717615416.8251 +MEM 592.386719 1717615416.9253 +MEM 599.863281 1717615417.0257 +MEM 600.121094 1717615417.1259 +MEM 603.472656 1717615417.2262 +MEM 606.523438 1717615417.3266 +MEM 606.523438 1717615417.4269 +MEM 606.523438 1717615417.5272 +MEM 606.523438 1717615417.6275 +MEM 606.523438 1717615417.7278 +MEM 606.523438 1717615417.8281 +MEM 606.523438 1717615417.9284 +MEM 606.523438 1717615418.0287 +MEM 606.523438 1717615418.1290 +MEM 606.523438 1717615418.2293 +MEM 606.523438 1717615418.3296 +MEM 606.523438 1717615418.4299 +MEM 606.523438 1717615418.5301 +MEM 606.523438 1717615418.6305 +MEM 606.523438 1717615418.7308 +MEM 606.523438 1717615418.8311 +MEM 606.523438 1717615418.9314 +MEM 606.523438 1717615419.0317 +MEM 606.523438 1717615419.1320 +MEM 606.523438 1717615419.2323 +MEM 606.523438 1717615419.3327 +MEM 606.523438 1717615419.4330 +MEM 606.523438 1717615419.5333 +MEM 606.523438 1717615419.6336 +MEM 606.523438 1717615419.7340 +MEM 606.523438 1717615419.8345 +MEM 606.523438 1717615419.9348 +MEM 606.523438 1717615420.0351 +MEM 606.523438 1717615420.1354 +MEM 606.523438 1717615420.2357 +MEM 606.523438 1717615420.3360 +MEM 606.523438 1717615420.4363 +MEM 606.523438 1717615420.5366 +MEM 606.523438 1717615420.6370 +MEM 606.523438 1717615420.7373 +MEM 606.523438 1717615420.8377 +MEM 606.523438 1717615420.9381 +MEM 606.523438 1717615421.0384 +MEM 606.523438 1717615421.1387 +MEM 606.523438 1717615421.2390 +MEM 606.523438 1717615421.3393 +MEM 606.523438 1717615421.4396 +MEM 606.523438 1717615421.5399 +MEM 606.523438 1717615421.6402 +MEM 606.523438 1717615421.7405 +MEM 606.523438 1717615421.8408 +MEM 606.523438 1717615421.9411 +MEM 606.523438 1717615422.0415 +MEM 606.523438 1717615422.1418 +MEM 612.191406 1717615422.2421 +MEM 612.191406 1717615422.3425 +MEM 612.707031 1717615422.4428 +MEM 618.636719 1717615422.5431 +MEM 618.636719 1717615422.6434 +MEM 625.597656 1717615422.7437 +MEM 625.597656 1717615422.8441 +MEM 631.785156 1717615422.9444 +MEM 632.816406 1717615423.0447 +MEM 633.074219 1717615423.1450 +MEM 640.292969 1717615423.2453 +MEM 640.292969 1717615423.3456 +MEM 647.769531 1717615423.4459 +MEM 647.769531 1717615423.5461 +MEM 654.214844 1717615423.6465 +MEM 655.246094 1717615423.7468 +MEM 655.761719 1717615423.8471 +MEM 662.722656 1717615423.9474 +MEM 662.980469 1717615424.0477 +MEM 670.199219 1717615424.1480 +MEM 670.457031 1717615424.2483 +MEM 677.160156 1717615424.3486 +MEM 677.933594 1717615424.4489 +MEM 678.449219 1717615424.5492 +MEM 685.410156 1717615424.6495 +MEM 685.410156 1717615424.7498 +MEM 692.886719 1717615424.8501 +MEM 692.886719 1717615424.9504 +MEM 700.363281 1717615425.0506 +MEM 700.363281 1717615425.1509 +MEM 700.878906 1717615425.2512 +MEM 707.839844 1717615425.3515 +MEM 708.097656 1717615425.4518 +MEM 652.933594 1717615425.5522 diff --git a/norch/__pycache__/tensor.cpython-38.pyc b/norch/__pycache__/tensor.cpython-38.pyc index ad61d69..c7cc044 100644 Binary files a/norch/__pycache__/tensor.cpython-38.pyc and b/norch/__pycache__/tensor.cpython-38.pyc differ diff --git a/norch/csrc/cuda.cu b/norch/csrc/cuda.cu index 268d8a2..3095130 100644 --- a/norch/csrc/cuda.cu +++ b/norch/csrc/cuda.cu @@ -24,9 +24,8 @@ __host__ void cpu_to_cuda(Tensor* tensor, int device_id) { tensor->data = data_tmp; - const char* device_str = "cuda"; - tensor->device = (char*)malloc(strlen(device_str) + 1); - strcpy(tensor->device, device_str); + tensor->device = (char*)malloc(strlen("cuda") + 1); + strcpy(tensor->device, "cuda"); } __host__ void cuda_to_cpu(Tensor* tensor) { @@ -37,9 +36,8 @@ __host__ void cuda_to_cpu(Tensor* tensor) { tensor->data = data_tmp; - const char* device_str = "cpu"; - tensor->device = (char*)malloc(strlen(device_str) + 1); - strcpy(tensor->device, device_str); + tensor->device = (char*)malloc(strlen("cpu") + 1); + strcpy(tensor->device, "cpu"); } __host__ void free_cuda(float* data) { diff --git a/norch/csrc/tensor.cpp b/norch/csrc/tensor.cpp index 54882bd..9443a6f 100644 --- a/norch/csrc/tensor.cpp +++ b/norch/csrc/tensor.cpp @@ -16,50 +16,40 @@ extern "C" { fprintf(stderr, "Memory allocation failed\n"); exit(1); } + tensor->data = data; + tensor->shape = shape; tensor->ndim = ndim; + tensor->device = (char*)malloc(strlen(device) + 1); + if (device != NULL) { + strcpy(tensor->device, device); + } else { + fprintf(stderr, "Memory allocation failed\n"); + exit(-1); + } + tensor->size = 1; for (int i = 0; i < ndim; i++) { tensor->size *= shape[i]; } - tensor->device = strdup(device); tensor->strides = (int*)malloc(ndim * sizeof(int)); - tensor->shape = (int*)malloc(ndim * sizeof(int)); - - if (tensor->device == NULL || tensor->strides == NULL || tensor->shape == NULL) { + if (tensor->strides == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } - - memcpy(tensor->shape, shape, tensor->ndim * sizeof(int)); - strcpy(tensor->device, device); - - if (strcmp(tensor->device, "cpu") == 0) { - tensor->data = (float*)malloc(tensor->size * sizeof(float)); - memcpy(tensor->data, data, tensor->size * sizeof(float)); - } else { - //already allocated on gpu - tensor->data = data; - } - int stride = 1; for (int i = ndim - 1; i >= 0; i--) { tensor->strides[i] = stride; stride *= shape[i]; } - + return tensor; } void delete_tensor(Tensor* tensor) { if (tensor != NULL) { - if (tensor->strides != NULL) { - free(tensor->strides); - tensor->strides = NULL; - } - if (tensor->shape != NULL) { free(tensor->shape); tensor->shape = NULL; @@ -88,6 +78,21 @@ extern "C" { } } + void delete_strides(Tensor* tensor) { + // as strides always are allocated within C code, it must be handled separatedly + if (tensor->strides != NULL) { + free(tensor->strides); + tensor->strides = NULL; + } + } + + void delete_device(Tensor* tensor) { + // as device always are allocated within C code, it must be handled separatedly + if (tensor->device != NULL) { + free(tensor->device); + tensor->device = NULL; + } + } float get_item(Tensor* tensor, int* indices) { int index = 0; @@ -96,10 +101,10 @@ extern "C" { } float result; - if (strcmp(tensor->device, "cuda") == 0) { - cudaMemcpy(&result, tensor->data + index, sizeof(float), cudaMemcpyDeviceToHost); - } else { + if (strcmp(tensor->device, "cpu") == 0) { result = tensor->data[index]; + } else { + cudaMemcpy(&result, tensor->data + index, sizeof(float), cudaMemcpyDeviceToHost); } return result; @@ -136,13 +141,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor1->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -158,21 +156,20 @@ extern "C" { shape[i] = tensor1->shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); - add_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor1->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } add_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); + add_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } @@ -206,13 +203,7 @@ extern "C" { broadcasted_size *= broadcasted_shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - float* result_data; - cudaMalloc((void **)&result_data, broadcasted_size * sizeof(float)); - add_broadcasted_tensor_cuda(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); - return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(broadcasted_size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); @@ -221,17 +212,16 @@ extern "C" { add_broadcasted_tensor_cpu(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, broadcasted_size * sizeof(float)); + add_broadcasted_tensor_cuda(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); + return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); } } Tensor* sum_tensor(Tensor* tensor, int axis, bool keepdim) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim; int* shape; @@ -259,36 +249,7 @@ extern "C" { axis_size *= shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - if (axis == -1) { - cudaMalloc((void**)&result_data, tensor->size * sizeof(float)); - } else { - cudaMalloc((void**)&result_data, axis_size * sizeof(float)); - } - sum_tensor_cuda(tensor, result_data, axis); - - if (keepdim) { - if (axis == -1){ - ndim = tensor->ndim; - shape = (int*) malloc((tensor->ndim) * sizeof(int)); - for (int i = 0; i < tensor->ndim; i++) { - shape[i] = 1; - } - } else { - shape = (int*) malloc((tensor->ndim) * sizeof(int)); - for (int i = 0; i < tensor->ndim; i++) { - shape[i] = tensor->shape[i]; - } - shape[axis] = 1; - ndim = tensor->ndim; - } - - } - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)calloc(axis_size, sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); @@ -314,18 +275,39 @@ extern "C" { } } - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + if (axis == -1) { + cudaMalloc((void**)&result_data, tensor->size * sizeof(float)); + } else { + cudaMalloc((void**)&result_data, axis_size * sizeof(float)); + } + sum_tensor_cuda(tensor, result_data, axis); + + if (keepdim) { + if (axis == -1){ + ndim = tensor->ndim; + shape = (int*) malloc((tensor->ndim) * sizeof(int)); + for (int i = 0; i < tensor->ndim; i++) { + shape[i] = 1; + } + } else { + shape = (int*) malloc((tensor->ndim) * sizeof(int)); + for (int i = 0; i < tensor->ndim; i++) { + shape[i] = tensor->shape[i]; + } + shape[axis] = 1; + ndim = tensor->ndim; + } + + } + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* max_tensor(Tensor* tensor, int axis, bool keepdim) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim; int* shape; if (axis == -1) { @@ -347,8 +329,35 @@ extern "C" { axis_size *= shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { + if (strcmp(tensor->device, "cpu") == 0) { + float* result_data = (float*)malloc(axis_size * sizeof(float)); + if (result_data == NULL) { + fprintf(stderr, "Memory allocation failed\n"); + exit(1); + } + max_tensor_cpu(tensor, result_data, axis_size, shape, axis); + if (keepdim) { + if (axis == -1){ + ndim = tensor->ndim; + shape = (int*) malloc((tensor->ndim) * sizeof(int)); + for (int i = 0; i < tensor->ndim; i++) { + shape[i] = 1; + } + } else { + shape = (int*) malloc((tensor->ndim) * sizeof(int)); + for (int i = 0; i < tensor->ndim; i++) { + shape[i] = tensor->shape[i]; + } + shape[axis] = 1; + ndim = tensor->ndim; + } + + } + return create_tensor(result_data, shape, ndim, tensor->device); + + } + else { float* result_data; if (axis == -1) { cudaMalloc((void**)&result_data, tensor->size * sizeof(float)); @@ -374,46 +383,11 @@ extern "C" { } } - return create_tensor(result_data, shape, ndim, device); - - } - else { - float* result_data = (float*)malloc(axis_size * sizeof(float)); - if (result_data == NULL) { - fprintf(stderr, "Memory allocation failed\n"); - exit(1); - } - max_tensor_cpu(tensor, result_data, axis_size, shape, axis); - - if (keepdim) { - if (axis == -1){ - ndim = tensor->ndim; - shape = (int*) malloc((tensor->ndim) * sizeof(int)); - for (int i = 0; i < tensor->ndim; i++) { - shape[i] = 1; - } - } else { - shape = (int*) malloc((tensor->ndim) * sizeof(int)); - for (int i = 0; i < tensor->ndim; i++) { - shape[i] = tensor->shape[i]; - } - shape[axis] = 1; - ndim = tensor->ndim; - } - - } - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* min_tensor(Tensor* tensor, int axis, bool keepdim) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim; int* shape; if (axis == -1) { @@ -435,8 +409,35 @@ extern "C" { axis_size *= shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { + if (strcmp(tensor->device, "cpu") == 0) { + float* result_data = (float*)malloc(axis_size * sizeof(float)); + if (result_data == NULL) { + fprintf(stderr, "Memory allocation failed\n"); + exit(1); + } + min_tensor_cpu(tensor, result_data, axis_size, shape, axis); + if (keepdim) { + if (axis == -1){ + ndim = tensor->ndim; + shape = (int*) malloc((tensor->ndim) * sizeof(int)); + for (int i = 0; i < tensor->ndim; i++) { + shape[i] = 1; + } + } else { + shape = (int*) malloc((tensor->ndim) * sizeof(int)); + for (int i = 0; i < tensor->ndim; i++) { + shape[i] = tensor->shape[i]; + } + shape[axis] = 1; + ndim = tensor->ndim; + } + + } + return create_tensor(result_data, shape, ndim, tensor->device); + + } + else { float* result_data; if (axis == -1) { cudaMalloc((void**)&result_data, tensor->size * sizeof(float)); @@ -462,35 +463,7 @@ extern "C" { } } - return create_tensor(result_data, shape, ndim, device); - - } - else { - float* result_data = (float*)malloc(axis_size * sizeof(float)); - if (result_data == NULL) { - fprintf(stderr, "Memory allocation failed\n"); - exit(1); - } - min_tensor_cpu(tensor, result_data, axis_size, shape, axis); - - if (keepdim) { - if (axis == -1){ - ndim = tensor->ndim; - shape = (int*) malloc((tensor->ndim) * sizeof(int)); - for (int i = 0; i < tensor->ndim; i++) { - shape[i] = 1; - } - } else { - shape = (int*) malloc((tensor->ndim) * sizeof(int)); - for (int i = 0; i < tensor->ndim; i++) { - shape[i] = tensor->shape[i]; - } - shape[axis] = 1; - ndim = tensor->ndim; - } - - } - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); } } @@ -505,13 +478,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor1->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -527,21 +493,20 @@ extern "C" { shape[i] = tensor1->shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); - sub_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor1->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } sub_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); + sub_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } @@ -575,13 +540,7 @@ extern "C" { broadcasted_size *= broadcasted_shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - float* result_data; - cudaMalloc((void **)&result_data, broadcasted_size * sizeof(float)); - sub_broadcasted_tensor_cuda(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); - return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(broadcasted_size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); @@ -590,6 +549,13 @@ extern "C" { sub_broadcasted_tensor_cpu(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); + } + else { + + float* result_data; + cudaMalloc((void **)&result_data, broadcasted_size * sizeof(float)); + sub_broadcasted_tensor_cuda(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); + return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); } } @@ -604,13 +570,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor1->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -626,33 +585,25 @@ extern "C" { shape[i] = tensor1->shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); - elementwise_mul_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor1->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } elementwise_mul_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); + elementwise_mul_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } Tensor* scalar_mul_tensor(Tensor* tensor, float scalar) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -664,33 +615,25 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - scalar_mul_tensor_cuda(tensor, scalar, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } scalar_mul_tensor_cpu(tensor, scalar, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + scalar_mul_tensor_cuda(tensor, scalar, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* scalar_div_tensor(float scalar, Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -702,33 +645,25 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - scalar_div_tensor_cuda(scalar, tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } scalar_div_tensor_cpu(scalar, tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + scalar_div_tensor_cuda(scalar, tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* tensor_div_scalar(Tensor* tensor, float scalar) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -740,21 +675,20 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - tensor_div_scalar_cuda(tensor, scalar, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } tensor_div_scalar_cpu(tensor, scalar, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + tensor_div_scalar_cuda(tensor, scalar, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } @@ -769,13 +703,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor1->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -791,21 +718,20 @@ extern "C" { shape[i] = tensor1->shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); - tensor_div_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor1->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } tensor_div_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); + tensor_div_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } @@ -822,13 +748,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor1->ndim + tensor2->ndim - 2; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -853,21 +772,20 @@ extern "C" { exit(1); } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, size * sizeof(float)); - matmul_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } matmul_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, size * sizeof(float)); + matmul_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } @@ -884,14 +802,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } - int ndim = 3; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -914,21 +824,20 @@ extern "C" { exit(1); } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, size * sizeof(float)); - broadcasted_batched_matmul_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } broadcasted_batched_matmul_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, size * sizeof(float)); + broadcasted_batched_matmul_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } @@ -952,14 +861,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } - int ndim = 3; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -982,34 +883,26 @@ extern "C" { exit(1); } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, size * sizeof(float)); - batched_matmul_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } batched_matmul_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, size * sizeof(float)); + batched_matmul_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } Tensor* tensor_pow_scalar(Tensor* tensor, float exponent) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1021,32 +914,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - tensor_pow_scalar_cuda(tensor, exponent, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } tensor_pow_scalar_cpu(tensor, exponent, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + tensor_pow_scalar_cuda(tensor, exponent, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* scalar_pow_tensor(float base, Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1058,32 +943,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - scalar_pow_tensor_cuda(base, tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } scalar_pow_tensor_cpu(base, tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + scalar_pow_tensor_cuda(base, tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* log_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1095,32 +972,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - log_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } log_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + log_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* reshape_tensor(Tensor* tensor, int* new_shape, int new_ndim) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = new_ndim; int* shape = (int*)malloc(ndim * sizeof(int)); @@ -1145,21 +1014,20 @@ extern "C" { exit(1); } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - assign_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } assign_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + assign_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } @@ -1174,13 +1042,6 @@ extern "C" { exit(1); } - char* device = (char*)malloc(strlen(tensor1->device) + 1); - if (device != NULL) { - strcpy(device, tensor1->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor1->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1196,21 +1057,20 @@ extern "C" { shape[i] = tensor1->shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); - equal_tensor_cuda(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor1->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } equal_tensor_cpu(tensor1, tensor2, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor1->size * sizeof(float)); + equal_tensor_cuda(tensor1, tensor2, result_data); + return create_tensor(result_data, shape, ndim, tensor1->device); } } @@ -1244,13 +1104,7 @@ extern "C" { broadcasted_size *= broadcasted_shape[i]; } - if (strcmp(tensor1->device, "cuda") == 0) { - float* result_data; - cudaMalloc((void **)&result_data, broadcasted_size * sizeof(float)); - equal_broadcasted_tensor_cuda(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); - return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); - } - else { + if (strcmp(tensor1->device, "cpu") == 0) { float* result_data = (float*)malloc(broadcasted_size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); @@ -1259,18 +1113,17 @@ extern "C" { equal_broadcasted_tensor_cpu(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, broadcasted_size * sizeof(float)); + equal_broadcasted_tensor_cuda(tensor1, tensor2, result_data, broadcasted_shape, broadcasted_size); + return create_tensor(result_data, broadcasted_shape, max_ndim, tensor1->device); } } Tensor* ones_like_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1282,32 +1135,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - ones_like_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } ones_like_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + ones_like_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* zeros_like_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1319,32 +1164,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - zeros_like_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } zeros_like_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + zeros_like_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* sin_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1356,32 +1193,25 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - sin_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } sin_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + sin_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* cos_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } + int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1393,32 +1223,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - cos_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } cos_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + cos_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* sigmoid_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1430,33 +1252,24 @@ extern "C" { shape[i] = tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - sigmoid_tensor_cuda(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); exit(1); } sigmoid_tensor_cpu(tensor, result_data); - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + sigmoid_tensor_cuda(tensor, result_data); + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* transpose_tensor(Tensor* tensor) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } - int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1470,27 +1283,7 @@ extern "C" { int size = tensor->size; - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, size * sizeof(float)); - switch (ndim) { - case 1: - transpose_1D_tensor_cuda(tensor, result_data); - break; - case 2: - transpose_2D_tensor_cuda(tensor, result_data); - break; - case 3: - transpose_3D_tensor_cuda(tensor, result_data); - break; - default: - fprintf(stderr, "Transpose only supports tensors up to 3 dimensions.\n"); - exit(-1); - } - return create_tensor(result_data, shape, ndim, device); - } - else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); @@ -1510,19 +1303,30 @@ extern "C" { fprintf(stderr, "Transpose only supports tensors up to 3 dimensions.\n"); exit(-1); } - return create_tensor(result_data, shape, ndim, device); + return create_tensor(result_data, shape, ndim, tensor->device); + } + else { + float* result_data; + cudaMalloc((void **)&result_data, size * sizeof(float)); + switch (ndim) { + case 1: + transpose_1D_tensor_cuda(tensor, result_data); + break; + case 2: + transpose_2D_tensor_cuda(tensor, result_data); + break; + case 3: + transpose_3D_tensor_cuda(tensor, result_data); + break; + default: + fprintf(stderr, "Transpose only supports tensors up to 3 dimensions.\n"); + exit(-1); + } + return create_tensor(result_data, shape, ndim, tensor->device); } } Tensor* transpose_axes_tensor(Tensor* tensor, int axis1, int axis2) { - char* device = (char*)malloc(strlen(tensor->device) + 1); - if (device != NULL) { - strcpy(device, tensor->device); - } else { - fprintf(stderr, "Memory allocation failed\n"); - exit(-1); - } - int ndim = tensor->ndim; int* shape = (int*)malloc(ndim * sizeof(int)); if (shape == NULL) { @@ -1539,13 +1343,15 @@ extern "C" { int size = tensor->size; - if (strcmp(tensor->device, "cuda") == 0) { - - float* result_data; - cudaMalloc((void **)&result_data, size * sizeof(float)); - assign_tensor_cuda(tensor, result_data); - - Tensor* new_tensor = create_tensor(result_data, shape, ndim, device); + if (strcmp(tensor->device, "cpu") == 0) { + float* result_data = (float*)malloc(size * sizeof(float)); + if (result_data == NULL) { + fprintf(stderr, "Memory allocation failed\n"); + exit(1); + } + assign_tensor_cpu(tensor, result_data); + + Tensor* new_tensor = create_tensor(result_data, shape, ndim, tensor->device); for (int i = 0; i < ndim; i++) { new_tensor->strides[i] = tensor->strides[i]; } @@ -1555,14 +1361,11 @@ extern "C" { return new_tensor; } else { - float* result_data = (float*)malloc(size * sizeof(float)); - if (result_data == NULL) { - fprintf(stderr, "Memory allocation failed\n"); - exit(1); - } - assign_tensor_cpu(tensor, result_data); - - Tensor* new_tensor = create_tensor(result_data, shape, ndim, device); + float* result_data; + cudaMalloc((void **)&result_data, size * sizeof(float)); + assign_tensor_cuda(tensor, result_data); + + Tensor* new_tensor = create_tensor(result_data, shape, ndim, tensor->device); for (int i = 0; i < ndim; i++) { new_tensor->strides[i] = tensor->strides[i]; } @@ -1587,17 +1390,16 @@ extern "C" { stride *= tensor->shape[i]; } - if (strcmp(tensor->device, "cuda") == 0) { - float* result_data; - cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); - make_contiguous_tensor_cuda(tensor, result_data, new_strides); - - } else { + if (strcmp(tensor->device, "cpu") == 0) { float* result_data = (float*)malloc(tensor->size * sizeof(float)); if (result_data == NULL) { fprintf(stderr, "Memory allocation failed\n"); } make_contiguous_tensor_cpu(tensor, result_data, new_strides); + } else { + float* result_data; + cudaMalloc((void **)&result_data, tensor->size * sizeof(float)); + make_contiguous_tensor_cuda(tensor, result_data, new_strides); } } } diff --git a/norch/csrc/tensor.h b/norch/csrc/tensor.h index 2465125..6793e10 100644 --- a/norch/csrc/tensor.h +++ b/norch/csrc/tensor.h @@ -13,6 +13,8 @@ typedef struct { extern "C" { Tensor* create_tensor(float* data, int* shape, int ndim, char* device); void delete_tensor(Tensor* tensor); + void delete_strides(Tensor* tensor); + void delete_device(Tensor* tensor); float get_item(Tensor* tensor, int* indices); Tensor* add_tensor(Tensor* tensor1, Tensor* tensor2); Tensor* sum_tensor(Tensor* tensor, int axis, bool keepdims); diff --git a/norch/libtensor.so b/norch/libtensor.so index 300caf1..598e064 100755 Binary files a/norch/libtensor.so and b/norch/libtensor.so differ diff --git a/norch/nn/__pycache__/loss.cpython-38.pyc b/norch/nn/__pycache__/loss.cpython-38.pyc index 26a0e8b..9f0eccb 100644 Binary files a/norch/nn/__pycache__/loss.cpython-38.pyc and b/norch/nn/__pycache__/loss.cpython-38.pyc differ diff --git a/norch/nn/loss.py b/norch/nn/loss.py index 6c0abea..a0be838 100644 --- a/norch/nn/loss.py +++ b/norch/nn/loss.py @@ -11,7 +11,7 @@ class Loss(Module, ABC): def forward(self, predictions, labels): raise NotImplementedError - + def __call__(self, *inputs): return self.forward(*inputs) diff --git a/norch/optim/optimizers/__pycache__/sgd.cpython-38.pyc b/norch/optim/optimizers/__pycache__/sgd.cpython-38.pyc index 6897844..34f2d4b 100644 Binary files a/norch/optim/optimizers/__pycache__/sgd.cpython-38.pyc and b/norch/optim/optimizers/__pycache__/sgd.cpython-38.pyc differ diff --git a/norch/optim/optimizers/sgd.py b/norch/optim/optimizers/sgd.py index 616d1b9..527f841 100644 --- a/norch/optim/optimizers/sgd.py +++ b/norch/optim/optimizers/sgd.py @@ -9,6 +9,7 @@ class SGD(Optimizer): self.momentum = momentum self._cache = {'velocity': [p.zeros_like() for (_, _, p) in self.parameters]} + def step(self): for i, (module, name, _) in enumerate(self.parameters): parameter = getattr(module, name) diff --git a/norch/tensor.py b/norch/tensor.py index 0287fb6..d2048b1 100644 --- a/norch/tensor.py +++ b/norch/tensor.py @@ -26,10 +26,10 @@ class Tensor: self.shape = shape.copy() - self.data_ctype = (ctypes.c_float * len(data))(*data.copy()) - self.shape_ctype = (ctypes.c_int * len(shape))(*shape.copy()) - self.ndim_ctype = ctypes.c_int(len(shape)) - self.device_ctype = device.encode('utf-8') + self._data_ctype = (ctypes.c_float * len(data))(*data.copy()) + self._shape_ctype = (ctypes.c_int * len(shape))(*shape.copy()) + self._ndim_ctype = ctypes.c_int(len(shape)) + self._device_ctype = device.encode('utf-8') self.ndim = len(shape) self.device = device @@ -47,15 +47,11 @@ class Tensor: Tensor._C.create_tensor.restype = ctypes.POINTER(CTensor) self.tensor = Tensor._C.create_tensor( - self.data_ctype, - self.shape_ctype, - self.ndim_ctype, - self.device_ctype + self._data_ctype, + self._shape_ctype, + self._ndim_ctype, + self._device_ctype ) - - del self.data_ctype - del self.shape_ctype - del self.device_ctype else: self.tensor = None, @@ -85,7 +81,21 @@ class Tensor: return flat_data, shape def __del__(self): - if self.tensor is not None: + + if hasattr(self, '_data_ctype') and self._data_ctype is not None: + # tensor created by user (ctypes) will be deleted by python garbage collector + # only strides need to be deallocated manually because it is created inside C code + Tensor._C.delete_strides.argtypes = [ctypes.POINTER(CTensor)] + Tensor._C.delete_strides.restype = None + Tensor._C.delete_strides(self.tensor) + + Tensor._C.delete_device.argtypes = [ctypes.POINTER(CTensor)] + Tensor._C.delete_device.restype = None + Tensor._C.delete_device(self.tensor) + + elif self.tensor is not None: + # tensor created during operations must be deallocated + Tensor._C.delete_tensor.argtypes = [ctypes.POINTER(CTensor)] Tensor._C.delete_tensor.restype = None Tensor._C.delete_tensor(self.tensor) diff --git a/profiling_results.prof b/profiling_results.prof new file mode 100644 index 0000000..1cc4be2 Binary files /dev/null and b/profiling_results.prof differ diff --git a/train_singlegpu.py b/train_singlegpu.py new file mode 100644 index 0000000..4d028eb --- /dev/null +++ b/train_singlegpu.py @@ -0,0 +1,97 @@ +import norch +import norch.nn as nn +import norch.optim as optim +from norch.norchvision import transforms as T +import random +random.seed(1) +from memory_profiler import profile + +@profile +def main(): + + + BATCH_SIZE = 32 + device = "cpu" + epochs = 10 + + transform = T.Compose( + [ + T.ToTensor(), + T.Reshape([-1, 784, 1]) + ] + ) + + target_transform = T.Compose( + [ + T.ToTensor() + ] + ) + + + print("Loading data") + train_data, test_data = norch.norchvision.datasets.MNIST.splits(transform=transform, target_transform=target_transform) + train_loader = norch.utils.data.DataLoader(train_data, batch_size=BATCH_SIZE) + + class MyModel(nn.Module): + def __init__(self): + super(MyModel, self).__init__() + self.fc1 = nn.Linear(784, 30) + self.sigmoid1 = nn.Sigmoid() + self.fc2 = nn.Linear(30, 10) + self.sigmoid2 = nn.Sigmoid() + + def forward(self, x): + out = self.fc1(x) + out = self.sigmoid1(out) + out = self.fc2(out) + out = self.sigmoid2(out) + + return out + + print("Creating model") + model = MyModel().to(device) + criterion = nn.CrossEntropyLoss() + optimizer = optim.SGD(model.parameters(), lr=0.01) + loss_list = [] + + print("Starting training") + for epoch in range(epochs): + + avg_loss = 0 + num_steps = 0 + + for idx, batch in enumerate(train_loader): + + if idx % 100 == 0 and idx > 0: + print(f"Epoch: {epoch}/{epochs} - Step: {idx} / {len(train_loader)}") + break + + inputs, target = batch + + inputs = inputs.to(device) + target = target.to(device) + + outputs = model(inputs) + loss = criterion(outputs, target) + + optimizer.zero_grad() + + loss.backward() + + optimizer.step() + + avg_loss += loss[0] + num_steps += 1 + + avg_loss = avg_loss / num_steps + print(f'Epoch [{epoch + 1}/{epochs}], Loss: {avg_loss:.4f}') + loss_list.append(avg_loss) + + break + + + +if __name__ == "__main__": + main() + +