Revert tensors quantization tree edits

2024-08-09 14:25:08 +02:00 · 2024-08-09 14:25:08 +02:00 · bd575f01de
commit bd575f01de
parent fc4ed23673
1 changed files with 111 additions and 143 deletions
--- a/src/llama.cpp
+++ b/src/llama.cpp
@ -15381,9 +15381,6 @@ static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type n
            }
        }
    } else if (name.find("attn_v.weight") != std::string::npos) {
        if (qs.params->attn_v_type < GGML_TYPE_COUNT) {
            new_type = qs.params->attn_v_type;
        } else {
        if      (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) {
            new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;
        }
@ -15420,12 +15417,8 @@ static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type n
            // TODO: explore better strategies
            new_type = GGML_TYPE_Q8_0;
        }
        }
        ++qs.i_attention_wv;
    } else if (name.find("attn_k.weight") != std::string::npos) {
        if (qs.params->attn_k_type < GGML_TYPE_COUNT) {
            new_type = qs.params->attn_k_type;
        } else {
        if (qs.model.hparams.n_expert == 8) {
            // for the 8-expert model, bumping this to Q8_0 trades just ~128MB
            // TODO: explore better strategies
@ -15437,24 +15430,16 @@ static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type n
        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {
            new_type = GGML_TYPE_IQ2_S;
        }
        }
    } else if (name.find("attn_q.weight") != std::string::npos) {
        if (qs.params->attn_q_type < GGML_TYPE_COUNT) {
            new_type = qs.params->attn_q_type;
        } else {
        if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {
            new_type = GGML_TYPE_IQ3_XXS;
        }
        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {
            new_type = GGML_TYPE_IQ2_S;
        }
        }
    } else if (name.find("ffn_down") != std::string::npos) {
        auto info = layer_info(qs.i_ffn_down, qs.n_ffn_down, name.c_str());
        int i_layer = info.first, n_layer = info.second;
        if (qs.params->ffn_down_type < GGML_TYPE_COUNT) {
            new_type = qs.params->ffn_down_type;
        } else {
        if      (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) new_type = GGML_TYPE_Q3_K;
        else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S) {
            if (i_layer < n_layer/8) new_type = GGML_TYPE_Q4_K;
@ -15496,12 +15481,8 @@ static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type n
            // same quantization as before imatrix stuff, and b) Q4_1/Q5_1 do go crazy on ffn_down without an imatrix.
            new_type = ftype == LLAMA_FTYPE_MOSTLY_Q4_0 ? GGML_TYPE_Q4_1 : GGML_TYPE_Q5_1;
        }
        }
        ++qs.i_ffn_down;
    } else if (name.find("attn_output.weight") != std::string::npos) {
        if (qs.params->attn_output_type < GGML_TYPE_COUNT) {
            new_type = qs.params->attn_output_type;
        } else {
        if (arch != LLM_ARCH_FALCON) {
            if (qs.model.hparams.n_expert == 8) {
                if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K   || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||
@ -15521,40 +15502,27 @@ static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type n
            if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q4_K;
        }
    }
    }
    else if (name.find("attn_qkv.weight") != std::string::npos) {
        if (qs.params->attn_qkv_type < GGML_TYPE_COUNT) {
            new_type = qs.params->attn_qkv_type;
        } else {
        if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L || ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {
            new_type = GGML_TYPE_Q4_K;
        }
        else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) new_type = GGML_TYPE_Q5_K;
        else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) new_type = GGML_TYPE_Q6_K;
    }
    }
    else if (name.find("ffn_gate") != std::string::npos) {
        auto info = layer_info(qs.i_ffn_gate, qs.n_ffn_gate, name.c_str());
        int i_layer = info.first, n_layer = info.second;
        if (qs.params->ffn_gate_type < GGML_TYPE_COUNT) {
            new_type = qs.params->ffn_gate_type;
        } else {
        if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {
            new_type = GGML_TYPE_IQ3_XXS;
        }
        }
        ++qs.i_ffn_gate;
    }
    else if (name.find("ffn_up") != std::string::npos) {
        auto info = layer_info(qs.i_ffn_up, qs.n_ffn_up, name.c_str());
        int i_layer = info.first, n_layer = info.second;
        if (qs.params->ffn_up_type < GGML_TYPE_COUNT) {
            new_type = qs.params->ffn_up_type;
        } else {
        if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {
            new_type = GGML_TYPE_IQ3_XXS;
        }
        }
        ++qs.i_ffn_up;
    }