llava-phi-3.5-mini-instruct_untrained / tokenizer_config.json

Upload processor

35f99a4 verified about 2 months ago

3.73 kB

	{
	"add_bos_token": false,
	"add_eos_token": false,
	"add_prefix_space": null,
	"added_tokens_decoder": {
	"0": {
	"content": "<unk>",
	"lstrip": false,
	"normalized": false,
	"rstrip": false,
	"single_word": false,
	"special": true
	},
	"1": {
	"content": "<s>",
	"lstrip": false,
	"normalized": false,
	"rstrip": false,
	"single_word": false,
	"special": true
	},
	"2": {
	"content": "</s>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": false
	},
	"32000": {
	"content": "<\|endoftext\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": false,
	"single_word": false,
	"special": true
	},
	"32001": {
	"content": "<\|assistant\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32002": {
	"content": "<\|placeholder1\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32003": {
	"content": "<\|placeholder2\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32004": {
	"content": "<\|placeholder3\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32005": {
	"content": "<\|placeholder4\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32006": {
	"content": "<\|system\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32007": {
	"content": "<\|end\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32008": {
	"content": "<\|placeholder5\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32009": {
	"content": "<\|placeholder6\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32010": {
	"content": "<\|user\|>",
	"lstrip": false,
	"normalized": false,
	"rstrip": true,
	"single_word": false,
	"special": true
	},
	"32011": {
	"content": "<image>",
	"lstrip": false,
	"normalized": false,
	"rstrip": false,
	"single_word": false,
	"special": true
	}
	},
	"bos_token": "<s>",
	"chat_template": "{% set is_splitted = index is defined and length is defined %}\n{% for message in messages %}\n {% set content = message['content'] + '<\|end\|>\\n' %}\n {% if message['role'] != 'assistant' or not is_splitted %}\n {% set content = '<\|' + message['role'] + '\|>\\n' + content %}\n {% endif %}\n {% if (is_splitted and index == 0 and loop.index0 == 0) or (not is_splitted and loop.index0 == 0) %}\n {% set content = bos_token + content %}\n {% endif %}\n {{- content -}}\n{% endfor %}\n{% if add_generation_prompt %}\n {{- '<\|assistant\|>\\n' -}}\n{% endif %}\n",
	"clean_up_tokenization_spaces": false,
	"eos_token": "<\|endoftext\|>",
	"legacy": false,
	"model_max_length": 131072,
	"pad_token": "<\|endoftext\|>",
	"padding_side": "left",
	"processor_class": "LlavaProcessor",
	"sp_model_kwargs": {},
	"tokenizer_class": "LlamaTokenizer",
	"unk_token": "<unk>",
	"use_default_system_prompt": false
	}