llama : cleanup unused mmq flags (#5772)

* cleanup unused --no-mul-mat-q,-nommq, -mmq, --mul-mat-q, mul_mat_q * remove: mul_mat_q in compare llama bench and usage * update llama-bench --------- Co-authored-by: slaren <slarengh@gmail.com>
2024-03-01 12:39:06 +01:00 · 2024-03-01 12:39:06 +01:00 · 3ab8b3a92e
commit 3ab8b3a92e
parent 9600d59e01
9 changed files with 10 additions and 56 deletions
--- a/common/common.h
+++ b/common/common.h
@ -115,7 +115,6 @@ struct gpt_params {

    bool   kl_divergence   = false; // compute KL-divergence

-    bool mul_mat_q         = true;  // if true, use mul_mat_q kernels instead of cuBLAS
    bool random_prompt     = false; // do not randomize prompt if none provided
    bool use_color         = false; // use color to distinguish generations and inputs
    bool interactive       = false; // interactive mode