llama : add gguf_remove_key + remove split meta during quantize (#6591)

* Remove split metadata when quantize model shards * Find metadata key by enum * Correct loop range for gguf_remove_key and code format * Free kv memory --------- Co-authored-by: z5269887 <z5269887@unsw.edu.au>
2024-04-12 18:45:06 +08:00 · 2024-04-12 18:45:06 +08:00 · 91c736015b
commit 91c736015b
parent 5c4d767ac0
3 changed files with 47 additions and 25 deletions
--- a/ggml.h
+++ b/ggml.h
@ -2289,6 +2289,9 @@ extern "C" {
    GGML_API char *         gguf_get_tensor_name  (const struct gguf_context * ctx, int i);
    GGML_API enum ggml_type gguf_get_tensor_type  (const struct gguf_context * ctx, int i);

+    // removes key if it exists
+    GGML_API void gguf_remove_key(struct gguf_context * ctx, const char * key);
+
    // overrides existing values or adds a new one
    GGML_API void gguf_set_val_u8  (struct gguf_context * ctx, const char * key, uint8_t  val);
    GGML_API void gguf_set_val_i8  (struct gguf_context * ctx, const char * key, int8_t   val);