llama : add pre-tokenizer regexes for BLOOM and gpt3-finnish (#8850)

This commit is contained in:
Esko Toivonen
2024-08-15 10:17:12 +03:00
committed by GitHub
parent d5492f0525
commit 6bda7ce6c3
5 changed files with 19 additions and 1 deletions

View File

@ -410,6 +410,8 @@ struct llm_tokenizer_bpe {
};
break;
case LLAMA_VOCAB_PRE_TYPE_PORO:
case LLAMA_VOCAB_PRE_TYPE_BLOOM:
case LLAMA_VOCAB_PRE_TYPE_GPT3_FINNISH:
regex_exprs = {
" ?[^(\\s|.,!?…。,、।۔،)]+",
};