tests : use Python to generate tokenizer tests for C++

2023-08-26 18:05:59 +03:00 · 2023-08-26 18:05:59 +03:00 · 70005bd5c9
commit 70005bd5c9
parent dfa058ef73
2 changed files with 46 additions and 35 deletions
--- a/tests/test-tokenizer-0.cpp
+++ b/tests/test-tokenizer-0.cpp
@ -6,6 +6,7 @@
 #include <map>
 #include <vector>

+// generate using test-tokenizer-0.py
 static const std::map<std::string, std::vector<llama_token>> & k_tests() {
    static std::map<std::string, std::vector<llama_token>> _k_tests = {
        { ""                      , {  }, },
@ -25,17 +26,8 @@ static const std::map<std::string, std::vector<llama_token>> & k_tests() {
        { " this is 🦙.cpp"        , {   29871,    445,    338,  29871,    243,    162,    169,    156,  29889,   8223, }, },
        { "w048 7tuijk dsdfhu"    , {     281,  29900,  29946,  29947,  29871,  29955,   9161,  13535,  18031,   2176,   6905, }, },
        { "нещо на Български"     , {    1538,   4851,    665,   1386,  29713,   1305, }, },
-        { "កាន់តែពិសេសអាចខលចេញ",
-                                    {  29871,  31849,  31324,  31934,    228,    162,    142,    228,    161,
-                                         146,    228,    162,    133,    228,    161,    153,    228,    161,    186,
-                                       31708,    228,    162,    132,  31708,    228,    161,    165,  31324,    228,
-                                         161,    136,    228,    161,    132,    228,    161,    158,    228,    161,
-                                         136,    228,    162,    132,    228,    161,    140, }, },
-        { "🚀 (normal) 😶‍🌫️ (multiple emojis concatenated) ✅ (only emoji that has its own token)",
-                                    {  29871,    243,    162,    157,    131,    313,   8945,  29897,  29871,
-                                         243,    162,    155,    185,  30722,    243,    162,    143,    174,  30598,
-                                         313,  20787,    953,   3848,    275,  16125,    630,  29897,  29871,  31681,
-                                         313,   6194,    953,  29877,   2397,    393,    756,    967,   1914,   5993,  29897, }, },
+        { "កាន់តែពិសេសអាចខលចេញ"   , {   29871,  31849,  31324,  31934,    228,    162,    142,    228,    161,    146,    228,    162,    133,    228,    161,    153,    228,    161,    186,  31708,    228,    162,    132,  31708,    228,    161,    165,  31324,    228,    161,    136,    228,    161,    132,    228,    161,    158,    228,    161,    136,    228,    162,    132,    228,    161,    140, }, },
+        { "🚀 (normal) 😶‍🌫️ (multiple emojis concatenated) ✅ (only emoji that has its own token)", {   29871,    243,    162,    157,    131,    313,   8945,  29897,  29871,    243,    162,    155,    185,  30722,    243,    162,    143,    174,  30598,    313,  20787,    953,   3848,    275,  16125,    630,  29897,  29871,  31681,    313,   6194,    953,  29877,   2397,    393,    756,    967,   1914,   5993,  29897, }, },
        { "Hello"                 , {   15043, }, },
        { " Hello"                , {   29871,  15043, }, },
        { "  Hello"               , {     259,  15043, }, },
@ -99,7 +91,14 @@ int main(int argc, char **argv) {
        const std::vector<llama_token> res_bos   = llama_tokenize(ctx, test_kv.first, true);
        const std::vector<llama_token> res_nobos = llama_tokenize(ctx, test_kv.first, false);

-        fprintf(stderr, "%s : '%s' tokenized to '%s'\n", __func__, test_kv.first.c_str(), llama_detokenize(ctx, res_bos).c_str());
+        printf("\n");
+        printf("src: '%s'\n", test_kv.first.c_str());
+        printf("res: '%s'\n", llama_detokenize(ctx, res_bos).c_str());
+        printf("tok: ");
+        for (const auto & tok : res_bos) {
+            printf("%d ", tok);
+        }
+        printf("\n");

        bool correct = res_nobos.size() == test_kv.second.size() && res_bos.size() == res_nobos.size() + 1 && res_bos[0] == 1;

--- a/tests/test-tokenizer-0.py
+++ b/tests/test-tokenizer-0.py
@ -56,3 +56,15 @@ print("'" + tokenizer.decode([15043]) + "'")        # 'Hello'
 print("'" + tokenizer.decode([15043, 15043]) + "'") # 'Hello Hello'
 print("'" + tokenizer.decode([29871, 15043]) + "'")               # ' Hello'
 print("'" + tokenizer.decode([29871, 15043, 29871, 15043]) + "'") # ' Hello  Hello'
+
+print("\n\ntests for C++:\n")
+for text in tests:
+    res = tokenizer.encode(text, add_bos=False)
+
+    k = text.replace('\n', '\\n')
+    k = k.replace('\t', '\\t')
+    k = '"' + k + '"'
+    print("{ %-24s, { " % k, end='')
+    for x in res:
+        print("%7d," % x, end='')
+    print(" }, },")