From 5c8eaec35963897b0f76b2cc78c78a25ad2fb4ff Mon Sep 17 00:00:00 2001 From: Andrej Karpathy Date: Thu, 18 Apr 2024 21:56:29 +0000 Subject: [PATCH] report all big mallocs --- train_gpt2.cu | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/train_gpt2.cu b/train_gpt2.cu index 7aafd38..e140dcf 100644 --- a/train_gpt2.cu +++ b/train_gpt2.cu @@ -1261,6 +1261,7 @@ void gpt2_build_from_checkpoint(GPT2 *model, const char* checkpoint_path) { // create memory for model parameters on the device model->params_memory = malloc_and_point_parameters(&model->params, model->param_sizes, 1); + printf("allocated %d MB for model parameters\n", (int)round(num_parameters * sizeof(float) / 1e6)); // read in all the parameters from file and copy them to device float* params_memory_cpu = (float*)mallocCheck(num_parameters * sizeof(float)); @@ -1350,9 +1351,9 @@ void gpt2_forward(GPT2 *model, int* inputs, int* targets, int B, int T) { for (size_t i = 0; i < NUM_ACTIVATION_TENSORS; i++) { num_activations += model->act_sizes[i]; } - printf("num_activations: %zu\n", num_activations); model->num_activations = num_activations; model->acts_memory = malloc_and_point_activations(&model->acts, model->act_sizes); + printf("allocated %d MB for activations\n", (int)round(num_activations * sizeof(float) / 1e6)); // also create memory for caching inputs and targets cudaCheck(cudaMalloc((void**)&model->inputs, B * T * sizeof(int))); cudaCheck(cudaMalloc((void**)&model->targets, B * T * sizeof(int))); @@ -1469,7 +1470,7 @@ void gpt2_backward(GPT2 *model) { if (model->grads_memory == NULL) { // allocate buffers for weight gradients model->grads_memory = malloc_and_point_parameters(&model->grads, model->param_sizes, 1); - + printf("allocated %d MB for parameter gradients\n", (int)round(model->num_parameters * sizeof(float) / 1e6)); // we're going to be clever for the activations backward pass. we don't need to exactly // mirror the forward pass acrtivations and we will save memory. size_t bw_act_sizes[NUM_ACTIVATION_TENSORS]; @@ -1493,6 +1494,8 @@ void gpt2_backward(GPT2 *model) { for (size_t i = 0; i < NUM_ACTIVATION_TENSORS; i++) { model->num_grad_acts += bw_act_sizes[i]; } + printf("allocated %d MB for activation gradients\n", (int)round(model->num_grad_acts * sizeof(float) / 1e6)); + // init gradients of parameters and activations to zero gpt2_zero_grad(model); } @@ -1602,6 +1605,8 @@ void gpt2_update(GPT2 *model, float learning_rate, float beta1, float beta2, flo cudaCheck(cudaMalloc((void**)&model->v_memory, model->num_parameters * sizeof(float))); cudaCheck(cudaMemset(model->m_memory, 0, model->num_parameters * sizeof(float))); cudaCheck(cudaMemset(model->v_memory, 0, model->num_parameters * sizeof(float))); + printf("allocated %d MB for AdamW optimizer state m\n", (int)round(model->num_parameters * sizeof(float) / 1e6)); + printf("allocated %d MB for AdamW optimizer state v\n", (int)round(model->num_parameters * sizeof(float) / 1e6)); } int block_size = 512;