Merge branch 'ggerganov:master' into bitnet

2024-06-10 10:51:47 +08:00 · 2024-06-10 10:51:47 +08:00 · 841c903ff9
commit 841c903ff9
parent abd798d70f 10ceba354a
218 changed files with 5021 additions and 8134 deletions
--- a/ggml.h
+++ b/ggml.h
@ -1467,7 +1467,6 @@ extern "C" {
    // rotary position embedding
    // if mode & 1 == 1, skip n_past elements (NOT SUPPORTED)
    // if mode & 2 == 1, GPT-NeoX style
-    // if mode & 4 == 1, ChatGLM style
    //
    // b is an int32 vector with size a->ne[2], it contains the positions
    // c is freq factors (e.g. phi3-128k), (optional)
@ -1476,8 +1475,7 @@ extern "C" {
            struct ggml_tensor  * a,
            struct ggml_tensor  * b,
            int                   n_dims,
-            int                   mode,
-            int                   n_ctx);
+            int                   mode);

    // in-place, returns view(a)
    GGML_API struct ggml_tensor * ggml_rope_inplace(
@ -1485,8 +1483,7 @@ extern "C" {
            struct ggml_tensor  * a,
            struct ggml_tensor  * b,
            int                   n_dims,
-            int                   mode,
-            int                   n_ctx);
+            int                   mode);

    // custom RoPE
    GGML_API struct ggml_tensor * ggml_rope_ext(
@ -1496,8 +1493,7 @@ extern "C" {
            struct ggml_tensor  * c,
            int                   n_dims,
            int                   mode,
-            int                   n_ctx,
-            int                   n_orig_ctx,
+            int                   n_ctx_orig,
            float                 freq_base,
            float                 freq_scale,
            float                 ext_factor,
@ -1513,8 +1509,7 @@ extern "C" {
            struct ggml_tensor  * c,
            int                   n_dims,
            int                   mode,
-            int                   n_ctx,
-            int                   n_orig_ctx,
+            int                   n_ctx_orig,
            float                 freq_base,
            float                 freq_scale,
            float                 ext_factor,
@ -1528,8 +1523,7 @@ extern "C" {
            struct ggml_tensor  * b,
            int                   n_dims,
            int                   mode,
-            int                   n_ctx,
-            int                   n_orig_ctx,
+            int                   n_ctx_orig,
            float                 freq_base,
            float                 freq_scale,
            float                 ext_factor,
@ -1544,8 +1538,7 @@ extern "C" {
            struct ggml_tensor  * b,
            int                   n_dims,
            int                   mode,
-            int                   n_ctx,
-            int                   n_orig_ctx,
+            int                   n_ctx_orig,
            float                 freq_base,
            float                 freq_scale,
            float                 ext_factor,
@ -1554,17 +1547,9 @@ extern "C" {
            float                 beta_slow),
        "use ggml_rope_ext_inplace instead");

-    struct ggml_tensor * ggml_rope_xpos_inplace(
-        struct ggml_context * ctx,
-        struct ggml_tensor  * a,
-        struct ggml_tensor  * b,
-        int                   n_dims,
-        float                 base,
-        bool                  down);
-
    // compute correction dims for YaRN RoPE scaling
    GGML_CALL void ggml_rope_yarn_corr_dims(
-        int n_dims, int n_orig_ctx, float freq_base, float beta_fast, float beta_slow, float dims[2]);
+        int n_dims, int n_ctx_orig, float freq_base, float beta_fast, float beta_slow, float dims[2]);

    // rotary position embedding backward, i.e compute dx from dy
    // a - dy
@ -1575,16 +1560,13 @@ extern "C" {
            struct ggml_tensor  * c,
            int                   n_dims,
            int                   mode,
-            int                   n_ctx,
-            int                   n_orig_ctx,
+            int                   n_ctx_orig,
            float                 freq_base,
            float                 freq_scale,
            float                 ext_factor,
            float                 attn_factor,
            float                 beta_fast,
-            float                 beta_slow,
-            float                 xpos_base,
-            bool                  xpos_down);
+            float                 beta_slow);

    // clamp
    // in-place, returns view(a)
@ -2427,7 +2409,6 @@ extern "C" {
    GGML_API int ggml_cpu_has_wasm_simd  (void);
    GGML_API int ggml_cpu_has_blas       (void);
    GGML_API int ggml_cpu_has_cuda       (void);
-    GGML_API int ggml_cpu_has_clblast    (void);
    GGML_API int ggml_cpu_has_vulkan     (void);
    GGML_API int ggml_cpu_has_kompute    (void);
    GGML_API int ggml_cpu_has_gpublas    (void);