llama : handle added special tokens like HF does

Now the BERT tokenizer actually uses the SEP and CLS tokens from
SpecialVocab.
This commit is contained in:
Jared Van Bortel 2024-03-27 16:59:49 -04:00
parent 748fc8baa3
commit 8803582721
3 changed files with 115 additions and 68 deletions

View file

@ -123,10 +123,10 @@ int main(int argc, char ** argv) {
inputs.push_back(inp);
}
// add eos if not present
// add SEP if not present
for (auto & inp : inputs) {
if (inp.empty() || inp.back() != llama_token_eos(model)) {
inp.push_back(llama_token_eos(model));
if (inp.empty() || inp.back() != llama_token_sep(model)) {
inp.push_back(llama_token_sep(model));
}
}