mirror of
				https://github.com/ggml-org/llama.cpp.git
				synced 2025-10-29 08:41:22 +00:00 
			
		
		
		
	llama : add pre-tokenizer regexes for BLOOM and gpt3-finnish (#8850)
This commit is contained in:
		| @@ -590,6 +590,12 @@ class Model: | ||||
|         if chkhsh == "855059429035d75a914d1eda9f10a876752e281a054a7a3d421ef0533e5b6249": | ||||
|             # ref: https://huggingface.co/HuggingFaceTB/SmolLM-135M | ||||
|             res = "smollm" | ||||
|         if chkhsh == "3c30d3ad1d6b64202cd222813e7736c2db6e1bd6d67197090fc1211fbc612ae7": | ||||
|             # ref: https://huggingface.co/bigscience/bloom | ||||
|             res = "bloom" | ||||
|         if chkhsh == "bc01ce58980e1db43859146dc51b1758b3b88729b217a74792e9f8d43e479d21": | ||||
|             # ref: https://huggingface.co/TurkuNLP/gpt3-finnish-small | ||||
|             res = "gpt3-finnish" | ||||
|  | ||||
|         if res is None: | ||||
|             logger.warning("\n") | ||||
| @@ -893,7 +899,7 @@ class GPTNeoXModel(Model): | ||||
|         return tensors | ||||
|  | ||||
|  | ||||
| @Model.register("BloomForCausalLM") | ||||
| @Model.register("BloomForCausalLM", "BloomModel") | ||||
| class BloomModel(Model): | ||||
|     model_arch = gguf.MODEL_ARCH.BLOOM | ||||
|  | ||||
|   | ||||
		Reference in New Issue
	
	Block a user
	 Esko Toivonen
					Esko Toivonen