From 64569135b25258cd283f8913ec1c63b7ae7588a7 Mon Sep 17 00:00:00 2001 From: Maxim Date: Tue, 28 Oct 2025 16:53:45 -0300 Subject: [PATCH] Remove unnecessary CUDA cache clearing - no memory leak reproducible now. --- model/kronos.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/model/kronos.py b/model/kronos.py index e95a853..5e1c336 100644 --- a/model/kronos.py +++ b/model/kronos.py @@ -432,8 +432,6 @@ def auto_regressive_inference(tokenizer, model, x, x_stamp, y_stamp, max_context x_token[0] = torch.cat([x_token[0], sample_pre], dim=1) x_token[1] = torch.cat([x_token[1], sample_post], dim=1) - torch.cuda.empty_cache() - input_tokens = [t[:, -max_context:].contiguous() for t in x_token] z = tokenizer.decode(input_tokens, half=True) z = z.reshape(batch_size, sample_count, z.size(1), z.size(2))