* llama : refactor GGUF constants into static maps * llama : check if model architecture is known * llama : refactor llama_model_load_internal() * gguf : add KV constant maps * llm : read arch-specific KVs * convert : add dummy scores + types * falcon : load tensor data (CPU only) * llama : fix loading progress bar * llama : add arch member to llama_model * falcon : CPU inference working * falcon : support non-40B models * falcon : minor * llama : minor updates ggml-ci * convert-falcon-hf-to-gguf.py : fix special token mapping * llama.cpp : llama default UNK token = id 0 * llama.cpp : fix bpe tokenizer * llama.cpp : fix the fix of bpe tokenizer * ggml : pass eps to ggml_norm * metal : implement RoPE (mode = 2) + avoid ggml_repeat * ggml : ggml_repeat always creates new tensor * falcon : copy-paste self-attention from LLaMA * metal : print extra compute pipeline info * falcon : minor changes (still chasing the Metal problem) * llama.cpp : fix linefeed token * metal : fix GELU kernel numerical stability by using precise::tanh * metal : temporary workaround for the concurrency optimization bug * falcon : add CUDA offloading (#2739) * llama : better model naming and size reporting * llama : prep new tokenizer support * llama : advanced BPE tokenizer based on ggllm.cpp imlpementation * llama : remove oboslete comment ggml-ci * common : remove obsolete BPE API + disable test-tokenizer-1 * llama : revert BPE special-case in llama_byte_to_token() * cuda : add TODOs for RoPE NeoX implementation * llama : default special tokens based on vocab type * perplexity : add log for start of tokenization --------- Co-authored-by: klosax <131523366+klosax@users.noreply.github.com> Co-authored-by: slaren <slarengh@gmail.com>
		
			
				
	
	
		
			37 lines
		
	
	
	
		
			1.9 KiB
		
	
	
	
		
			CMake
		
	
	
	
	
	
			
		
		
	
	
			37 lines
		
	
	
	
		
			1.9 KiB
		
	
	
	
		
			CMake
		
	
	
	
	
	
| function(llama_build_executable source)
 | |
|     get_filename_component(TEST_TARGET ${source} NAME_WE)
 | |
|     add_executable(${TEST_TARGET} ${source})
 | |
|     install(TARGETS ${TEST_TARGET} RUNTIME)
 | |
|     target_link_libraries(${TEST_TARGET} PRIVATE llama common)
 | |
| endfunction()
 | |
| 
 | |
| function(llama_test_executable name source)
 | |
|     get_filename_component(TEST_TARGET ${source} NAME_WE)
 | |
|     # add_executable(${TEST_TARGET} ${source})
 | |
|     # install(TARGETS ${TEST_TARGET} RUNTIME)
 | |
|     # target_link_libraries(${TEST_TARGET} PRIVATE llama)
 | |
|     add_test(NAME ${name} COMMAND $<TARGET_FILE:${TEST_TARGET}> ${ARGN})
 | |
| endfunction()
 | |
| 
 | |
| function(llama_build_and_test_executable source)
 | |
|     get_filename_component(TEST_TARGET ${source} NAME_WE)
 | |
|     add_executable(${TEST_TARGET} ${source})
 | |
|     install(TARGETS ${TEST_TARGET} RUNTIME)
 | |
|     target_link_libraries(${TEST_TARGET} PRIVATE llama common)
 | |
|     add_test(NAME ${TEST_TARGET} COMMAND $<TARGET_FILE:${TEST_TARGET}> ${ARGN})
 | |
| endfunction()
 | |
| 
 | |
| # llama_build_and_test_executable(test-double-float.cpp) # SLOW
 | |
| llama_build_and_test_executable(test-quantize-fns.cpp)
 | |
| llama_build_and_test_executable(test-quantize-perf.cpp)
 | |
| llama_build_and_test_executable(test-sampling.cpp)
 | |
| llama_build_executable(test-tokenizer-0.cpp)
 | |
| llama_test_executable (test-tokenizer-0.llama test-tokenizer-0.cpp ${CMAKE_CURRENT_SOURCE_DIR}/../models/ggml-vocab-llama.gguf)
 | |
| llama_build_executable(test-tokenizer-1.cpp)
 | |
| # test-tokenizer-1 requires a BPE vocab. re-enable when we have one.
 | |
| #llama_test_executable (test-tokenizer-1.llama test-tokenizer-1.cpp ${CMAKE_CURRENT_SOURCE_DIR}/../models/ggml-vocab-falcon.gguf)
 | |
| #llama_test_executable(test-tokenizer-1.aquila test-tokenizer-1.cpp ${CMAKE_CURRENT_SOURCE_DIR}/../models/ggml-vocab-aquila.gguf)
 | |
| llama_build_and_test_executable(test-grammar-parser.cpp)
 | |
| llama_build_and_test_executable(test-llama-grammar.cpp)
 | |
| llama_build_and_test_executable(test-grad0.cpp) # SLOW
 | |
| # llama_build_and_test_executable(test-opt.cpp) # SLOW
 |