From fb0471d1753824e75474c24f82fbdd54c94dceda Mon Sep 17 00:00:00 2001
From: pockers21 <134406831+pockers21@users.noreply.github.com>
Date: Mon, 28 Apr 2025 06:45:40 -0700
Subject: [PATCH] context : do not clear output buffer on reserve (#13152)

Co-authored-by: pockers21 <liyang2@uniontech.com>
---
 src/llama-context.cpp | 2 --
 1 file changed, 2 deletions(-)

diff --git a/src/llama-context.cpp b/src/llama-context.cpp
index a52b6850..e49225aa 100644
--- a/src/llama-context.cpp
+++ b/src/llama-context.cpp
@@ -1536,8 +1536,6 @@ int32_t llama_context::output_reserve(int32_t n_outputs) {
     // set all ids as invalid (negative)
     std::fill(output_ids.begin(), output_ids.end(), -1);
 
-    ggml_backend_buffer_clear(buf_output.get(), 0);
-
     this->n_outputs     = 0;
     this->n_outputs_max = n_outputs_max;