llama : remove all_pos_0, all_pos_1, all_seq_id from llama_batch (#9745)

* refactor llama_batch_get_one * adapt all examples * fix simple.cpp * fix llama_bench * fix * fix context shifting * free batch before return * use common_batch_add, reuse llama_batch in loop * null terminated seq_id list * fix save-load-state example * fix perplexity * correct token pos in llama_batch_allocr
2024-10-18 23:18:01 +02:00 · 2024-10-18 23:18:01 +02:00 · cda0e4b648
commit cda0e4b648
parent afd9909a64
22 changed files with 205 additions and 118 deletions
--- a/examples/simple/simple.cpp
+++ b/examples/simple/simple.cpp
@ -138,7 +138,7 @@ int main(int argc, char ** argv) {

    // prepare a batch for the prompt

-    llama_batch batch = llama_batch_get_one(prompt_tokens.data(), prompt_tokens.size(), 0, 0);
+    llama_batch batch = llama_batch_get_one(prompt_tokens.data(), prompt_tokens.size());

    // main loop

@ -175,7 +175,7 @@ int main(int argc, char ** argv) {
            fflush(stdout);

            // prepare the next batch with the sampled token
-            batch = llama_batch_get_one(&new_token_id, 1, n_pos, 0);
+            batch = llama_batch_get_one(&new_token_id, 1);

            n_decode += 1;
        }