Remove unnecessary CUDA cache clearing - no memory leak reproducible now.

This commit is contained in:
Maxim
2025-10-29 12:04:43 -03:00
parent 21eb36afc7
commit 64569135b2
-2
View File
@@ -432,8 +432,6 @@ def auto_regressive_inference(tokenizer, model, x, x_stamp, y_stamp, max_context
x_token[0] = torch.cat([x_token[0], sample_pre], dim=1)
x_token[1] = torch.cat([x_token[1], sample_post], dim=1)
torch.cuda.empty_cache()
input_tokens = [t[:, -max_context:].contiguous() for t in x_token]
z = tokenizer.decode(input_tokens, half=True)
z = z.reshape(batch_size, sample_count, z.size(1), z.size(2))