llama : refactor src/llama.cpp (#10902)

* llama : scatter llama.cpp into multiple modules (wip) * llama : control-vector -> adapter * llama : arch * llama : mmap ggml-ci * ci : remove BUILD_SHARED_LIBS=OFF ggml-ci * llama : arch (cont) ggml-ci * llama : chat ggml-ci * llama : model ggml-ci * llama : hparams ggml-ci * llama : adapter ggml-ci * examples : fix ggml-ci * rebase ggml-ci * minor * llama : kv cache ggml-ci * llama : impl ggml-ci * llama : batch ggml-ci * cont ggml-ci * llama : context ggml-ci * minor * llama : context (cont) ggml-ci * llama : model loader ggml-ci * common : update lora ggml-ci * llama : quant ggml-ci * llama : quant (cont) ggml-ci * minor [no ci]
2025-01-03 10:18:53 +02:00 · 2025-01-03 10:18:53 +02:00 · f66f582927
commit f66f582927
parent 2f0ee84b9b
61 changed files with 12193 additions and 11649 deletions
--- a/examples/main/main.cpp
+++ b/examples/main/main.cpp
@ -145,18 +145,18 @@ int main(int argc, char ** argv) {
    llama_context * ctx = nullptr;
    common_sampler * smpl = nullptr;

-    std::vector<common_chat_msg> chat_msgs;
-
    g_model = &model;
    g_ctx = &ctx;
    g_smpl = &smpl;

+    std::vector<common_chat_msg> chat_msgs;
+
    // load the model and apply lora adapter, if any
    LOG_INF("%s: load the model and apply lora adapter, if any\n", __func__);
    common_init_result llama_init = common_init_from_params(params);

-    model = llama_init.model;
-    ctx = llama_init.context;
+    model = llama_init.model.get();
+    ctx = llama_init.context.get();

    if (model == NULL) {
        LOG_ERR("%s: error: unable to load model\n", __func__);
@ -889,9 +889,6 @@ int main(int argc, char ** argv) {

    common_sampler_free(smpl);

-    llama_free(ctx);
-    llama_free_model(model);
-
    llama_backend_free();

    ggml_threadpool_free_fn(threadpool);