PR #4766

2024-01-10 11:29:04 -05:00 · 2024-01-10 11:29:04 -05:00 · 1eb8804c18
commit 1eb8804c18
parent 3773e1afe7
18 changed files with 2183 additions and 2081 deletions
--- a/common/common.h
+++ b/common/common.h
@ -59,6 +59,7 @@ struct gpt_params {
    float   p_split                         = 0.1f;  // speculative decoding split probability
    int32_t n_gpu_layers                    = -1;    // number of layers to store in VRAM (-1 - use default)
    int32_t n_gpu_layers_draft              = -1;    // number of layers to store in VRAM for the draft model (-1 - use default)
+    llama_split_mode split_mode             = LLAMA_SPLIT_LAYER; // how to split the model across GPUs
    int32_t main_gpu                        = 0;     // the GPU that is used for scratch and small tensors
    float   tensor_split[LLAMA_MAX_DEVICES] = {0};   // how split tensors should be distributed across GPUs
    int32_t n_beams                         = 0;     // if non-zero then use beam search of given width.