nllb-200-600M-lez-rus-azj / tokenizer_config.json
AlidarAsvarov's picture
Upload 7 files
23d77b1 verified
{
"added_tokens_decoder": {
"0": {
"content": "<s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"1": {
"content": "<pad>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"2": {
"content": "</s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"3": {
"content": "<unk>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269092": {
"content": "<mask>",
"lstrip": true,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": true
},
"269093": {
"content": "ace_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269094": {
"content": "ace_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269095": {
"content": "acm_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269096": {
"content": "acq_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269097": {
"content": "aeb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269098": {
"content": "afr_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269099": {
"content": "ajp_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269100": {
"content": "aka_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269101": {
"content": "als_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269102": {
"content": "amh_Ethi",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269103": {
"content": "apc_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269104": {
"content": "arb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269105": {
"content": "ars_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269106": {
"content": "ary_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269107": {
"content": "arz_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269108": {
"content": "asm_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269109": {
"content": "ast_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269110": {
"content": "awa_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269111": {
"content": "ayr_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269112": {
"content": "azb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269113": {
"content": "azj_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269114": {
"content": "bak_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269115": {
"content": "bam_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269116": {
"content": "ban_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269117": {
"content": "bel_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269118": {
"content": "bem_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269119": {
"content": "ben_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269120": {
"content": "bho_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269121": {
"content": "bjn_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269122": {
"content": "bjn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269123": {
"content": "bod_Tibt",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269124": {
"content": "bos_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269125": {
"content": "bug_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269126": {
"content": "bul_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269127": {
"content": "cat_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269128": {
"content": "ceb_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269129": {
"content": "ces_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269130": {
"content": "cjk_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269131": {
"content": "ckb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269132": {
"content": "crh_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269133": {
"content": "cym_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269134": {
"content": "dan_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269135": {
"content": "deu_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269136": {
"content": "dik_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269137": {
"content": "dyu_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269138": {
"content": "dzo_Tibt",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269139": {
"content": "ell_Grek",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269140": {
"content": "eng_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269141": {
"content": "epo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269142": {
"content": "est_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269143": {
"content": "eus_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269144": {
"content": "ewe_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269145": {
"content": "fao_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269146": {
"content": "fij_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269147": {
"content": "fin_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269148": {
"content": "fon_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269149": {
"content": "fra_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269150": {
"content": "fur_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269151": {
"content": "fuv_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269152": {
"content": "gaz_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269153": {
"content": "gla_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269154": {
"content": "gle_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269155": {
"content": "glg_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269156": {
"content": "grn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269157": {
"content": "guj_Gujr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269158": {
"content": "hat_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269159": {
"content": "hau_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269160": {
"content": "heb_Hebr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269161": {
"content": "hin_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269162": {
"content": "hne_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269163": {
"content": "hrv_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269164": {
"content": "hun_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269165": {
"content": "hye_Armn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269166": {
"content": "ibo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269167": {
"content": "ilo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269168": {
"content": "ind_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269169": {
"content": "isl_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269170": {
"content": "ita_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269171": {
"content": "jav_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269172": {
"content": "jpn_Jpan",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269173": {
"content": "kab_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269174": {
"content": "kac_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269175": {
"content": "kam_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269176": {
"content": "kan_Knda",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269177": {
"content": "kas_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269178": {
"content": "kas_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269179": {
"content": "kat_Geor",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269180": {
"content": "kaz_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269181": {
"content": "kbp_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269182": {
"content": "kea_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269183": {
"content": "khk_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269184": {
"content": "khm_Khmr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269185": {
"content": "kik_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269186": {
"content": "kin_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269187": {
"content": "kir_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269188": {
"content": "kmb_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269189": {
"content": "kmr_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269190": {
"content": "knc_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269191": {
"content": "knc_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269192": {
"content": "kon_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269193": {
"content": "kor_Hang",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269194": {
"content": "lao_Laoo",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269195": {
"content": "lez_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269196": {
"content": "lij_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269197": {
"content": "lim_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269198": {
"content": "lin_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269199": {
"content": "lit_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269200": {
"content": "lmo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269201": {
"content": "ltg_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269202": {
"content": "ltz_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269203": {
"content": "lua_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269204": {
"content": "lug_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269205": {
"content": "luo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269206": {
"content": "lus_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269207": {
"content": "lvs_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269208": {
"content": "mag_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269209": {
"content": "mai_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269210": {
"content": "mal_Mlym",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269211": {
"content": "mar_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269212": {
"content": "min_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269213": {
"content": "mkd_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269214": {
"content": "mlt_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269215": {
"content": "mni_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269216": {
"content": "mos_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269217": {
"content": "mri_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269218": {
"content": "mya_Mymr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269219": {
"content": "nld_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269220": {
"content": "nno_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269221": {
"content": "nob_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269222": {
"content": "npi_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269223": {
"content": "nso_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269224": {
"content": "nus_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269225": {
"content": "nya_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269226": {
"content": "oci_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269227": {
"content": "ory_Orya",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269228": {
"content": "pag_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269229": {
"content": "pan_Guru",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269230": {
"content": "pap_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269231": {
"content": "pbt_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269232": {
"content": "pes_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269233": {
"content": "plt_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269234": {
"content": "pol_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269235": {
"content": "por_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269236": {
"content": "prs_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269237": {
"content": "quy_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269238": {
"content": "ron_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269239": {
"content": "run_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269240": {
"content": "rus_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269241": {
"content": "sag_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269242": {
"content": "san_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269243": {
"content": "sat_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269244": {
"content": "scn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269245": {
"content": "shn_Mymr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269246": {
"content": "sin_Sinh",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269247": {
"content": "slk_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269248": {
"content": "slv_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269249": {
"content": "smo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269250": {
"content": "sna_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269251": {
"content": "snd_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269252": {
"content": "som_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269253": {
"content": "sot_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269254": {
"content": "spa_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269255": {
"content": "srd_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269256": {
"content": "srp_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269257": {
"content": "ssw_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269258": {
"content": "sun_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269259": {
"content": "swe_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269260": {
"content": "swh_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269261": {
"content": "szl_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269262": {
"content": "tam_Taml",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269263": {
"content": "taq_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269264": {
"content": "taq_Tfng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269265": {
"content": "tat_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269266": {
"content": "tel_Telu",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269267": {
"content": "tgk_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269268": {
"content": "tgl_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269269": {
"content": "tha_Thai",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269270": {
"content": "tir_Ethi",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269271": {
"content": "tpi_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269272": {
"content": "tsn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269273": {
"content": "tso_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269274": {
"content": "tuk_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269275": {
"content": "tum_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269276": {
"content": "tur_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269277": {
"content": "twi_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269278": {
"content": "tzm_Tfng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269279": {
"content": "uig_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269280": {
"content": "ukr_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269281": {
"content": "umb_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269282": {
"content": "urd_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269283": {
"content": "uzn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269284": {
"content": "vec_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269285": {
"content": "vie_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269286": {
"content": "war_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269287": {
"content": "wol_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269288": {
"content": "xho_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269289": {
"content": "ydd_Hebr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269290": {
"content": "yor_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269291": {
"content": "yue_Hant",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269292": {
"content": "zho_Hans",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269293": {
"content": "zho_Hant",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269294": {
"content": "zsm_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269295": {
"content": "zul_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
}
},
"additional_special_tokens": [
"ace_Arab",
"ace_Latn",
"acm_Arab",
"acq_Arab",
"aeb_Arab",
"afr_Latn",
"ajp_Arab",
"aka_Latn",
"als_Latn",
"amh_Ethi",
"apc_Arab",
"arb_Arab",
"ars_Arab",
"ary_Arab",
"arz_Arab",
"asm_Beng",
"ast_Latn",
"awa_Deva",
"ayr_Latn",
"azb_Arab",
"azj_Latn",
"bak_Cyrl",
"bam_Latn",
"ban_Latn",
"bel_Cyrl",
"bem_Latn",
"ben_Beng",
"bho_Deva",
"bjn_Arab",
"bjn_Latn",
"bod_Tibt",
"bos_Latn",
"bug_Latn",
"bul_Cyrl",
"cat_Latn",
"ceb_Latn",
"ces_Latn",
"cjk_Latn",
"ckb_Arab",
"crh_Latn",
"cym_Latn",
"dan_Latn",
"deu_Latn",
"dik_Latn",
"dyu_Latn",
"dzo_Tibt",
"ell_Grek",
"eng_Latn",
"epo_Latn",
"est_Latn",
"eus_Latn",
"ewe_Latn",
"fao_Latn",
"fij_Latn",
"fin_Latn",
"fon_Latn",
"fra_Latn",
"fur_Latn",
"fuv_Latn",
"gaz_Latn",
"gla_Latn",
"gle_Latn",
"glg_Latn",
"grn_Latn",
"guj_Gujr",
"hat_Latn",
"hau_Latn",
"heb_Hebr",
"hin_Deva",
"hne_Deva",
"hrv_Latn",
"hun_Latn",
"hye_Armn",
"ibo_Latn",
"ilo_Latn",
"ind_Latn",
"isl_Latn",
"ita_Latn",
"jav_Latn",
"jpn_Jpan",
"kab_Latn",
"kac_Latn",
"kam_Latn",
"kan_Knda",
"kas_Arab",
"kas_Deva",
"kat_Geor",
"kaz_Cyrl",
"kbp_Latn",
"kea_Latn",
"khk_Cyrl",
"khm_Khmr",
"kik_Latn",
"kin_Latn",
"kir_Cyrl",
"kmb_Latn",
"kmr_Latn",
"knc_Arab",
"knc_Latn",
"kon_Latn",
"kor_Hang",
"lao_Laoo",
"lez_Cyrl",
"lij_Latn",
"lim_Latn",
"lin_Latn",
"lit_Latn",
"lmo_Latn",
"ltg_Latn",
"ltz_Latn",
"lua_Latn",
"lug_Latn",
"luo_Latn",
"lus_Latn",
"lvs_Latn",
"mag_Deva",
"mai_Deva",
"mal_Mlym",
"mar_Deva",
"min_Latn",
"mkd_Cyrl",
"mlt_Latn",
"mni_Beng",
"mos_Latn",
"mri_Latn",
"mya_Mymr",
"nld_Latn",
"nno_Latn",
"nob_Latn",
"npi_Deva",
"nso_Latn",
"nus_Latn",
"nya_Latn",
"oci_Latn",
"ory_Orya",
"pag_Latn",
"pan_Guru",
"pap_Latn",
"pbt_Arab",
"pes_Arab",
"plt_Latn",
"pol_Latn",
"por_Latn",
"prs_Arab",
"quy_Latn",
"ron_Latn",
"run_Latn",
"rus_Cyrl",
"sag_Latn",
"san_Deva",
"sat_Beng",
"scn_Latn",
"shn_Mymr",
"sin_Sinh",
"slk_Latn",
"slv_Latn",
"smo_Latn",
"sna_Latn",
"snd_Arab",
"som_Latn",
"sot_Latn",
"spa_Latn",
"srd_Latn",
"srp_Cyrl",
"ssw_Latn",
"sun_Latn",
"swe_Latn",
"swh_Latn",
"szl_Latn",
"tam_Taml",
"taq_Latn",
"taq_Tfng",
"tat_Cyrl",
"tel_Telu",
"tgk_Cyrl",
"tgl_Latn",
"tha_Thai",
"tir_Ethi",
"tpi_Latn",
"tsn_Latn",
"tso_Latn",
"tuk_Latn",
"tum_Latn",
"tur_Latn",
"twi_Latn",
"tzm_Tfng",
"uig_Arab",
"ukr_Cyrl",
"umb_Latn",
"urd_Arab",
"uzn_Latn",
"vec_Latn",
"vie_Latn",
"war_Latn",
"wol_Latn",
"xho_Latn",
"ydd_Hebr",
"yor_Latn",
"yue_Hant",
"zho_Hans",
"zho_Hant",
"zsm_Latn",
"zul_Latn"
],
"bos_token": "<s>",
"clean_up_tokenization_spaces": false,
"cls_token": "<s>",
"eos_token": "</s>",
"legacy_behaviour": false,
"mask_token": "<mask>",
"model_max_length": 1024,
"pad_token": "<pad>",
"sep_token": "</s>",
"sp_model_kwargs": {},
"src_lang": "azj_Latn",
"tgt_lang": null,
"tokenizer_class": "NllbTokenizer",
"unk_token": "<unk>"
}