perplexity : fix ETA by warming up the model with an empty run

This commit is contained in:
Georgi Gerganov 2023-09-03 13:42:56 +03:00
parent 6519e9c99c
commit 8f429fa511
No known key found for this signature in database
GPG key ID: 449E073F9DC10735
2 changed files with 8 additions and 8 deletions

View file

@ -752,6 +752,14 @@ std::tuple<struct llama_model *, struct llama_context *> llama_init_from_gpt_par
params.logit_bias[llama_token_eos(lctx)] = -INFINITY;
}
{
LOG("warming up the model with an empty run\n");
const std::vector<llama_token> tmp = { llama_token_bos(lctx), };
llama_eval(lctx, tmp.data(), tmp.size(), 0, params.n_threads);
llama_reset_timings(lctx);
}
return std::make_tuple(model, lctx);
}