Remove usless check in llama-quantize (#2394)

Co-authored-by: Iwan Kawrakow <iwan.kawrakow@gmail.com>
This commit is contained in:
Kawrakow 2026-09-02 18:45:49 +02:00 committed by GitHub
parent c2206b80da
commit e560283754
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
1 changed files with 6 additions and 6 deletions

View File

@ -1286,12 +1286,12 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
// - qs.n_attention_wv == 3 * model.hparams.n_layer for Encoder-Decoder models
// - model.arch == LLM_ARCH_DECI for Deci-Nemotron models
//
GGML_ASSERT((qs.n_attention_wv == 0 ||
qs.n_attention_wv == (int)model.hparams.n_layer ||
qs.n_attention_wv == 3 * (int)model.hparams.n_layer ||
model.arch == LLM_ARCH_DECI ||
model.arch == LLM_ARCH_GEMMA4 ||
model.arch == LLM_ARCH_UNKNOWN) && "n_attention_wv is unexpected");
//GGML_ASSERT((qs.n_attention_wv == 0 ||
// qs.n_attention_wv == (int)model.hparams.n_layer ||
// qs.n_attention_wv == 3 * (int)model.hparams.n_layer ||
// model.arch == LLM_ARCH_DECI ||
// model.arch == LLM_ARCH_GEMMA4 ||
// model.arch == LLM_ARCH_UNKNOWN) && "n_attention_wv is unexpected");
size_t total_size_org = 0;
size_t total_size_new = 0;