dilmash-til / tokenizer_config.json
murodbek's picture
uploading tokenizer
0c514c7 verified
raw
history blame contribute delete
No virus
40.1 kB
{
"added_tokens_decoder": {
"0": {
"content": "<s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"1": {
"content": "<pad>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"2": {
"content": "</s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"3": {
"content": "<unk>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269195": {
"content": "ace_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269196": {
"content": "ace_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269197": {
"content": "acm_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269198": {
"content": "acq_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269199": {
"content": "aeb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269200": {
"content": "afr_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269201": {
"content": "ajp_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269202": {
"content": "aka_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269203": {
"content": "amh_Ethi",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269204": {
"content": "apc_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269205": {
"content": "arb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269206": {
"content": "ars_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269207": {
"content": "ary_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269208": {
"content": "arz_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269209": {
"content": "asm_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269210": {
"content": "ast_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269211": {
"content": "awa_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269212": {
"content": "ayr_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269213": {
"content": "azb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269214": {
"content": "azj_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269215": {
"content": "bak_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269216": {
"content": "bam_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269217": {
"content": "ban_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269218": {
"content": "bel_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269219": {
"content": "bem_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269220": {
"content": "ben_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269221": {
"content": "bho_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269222": {
"content": "bjn_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269223": {
"content": "bjn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269224": {
"content": "bod_Tibt",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269225": {
"content": "bos_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269226": {
"content": "bug_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269227": {
"content": "bul_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269228": {
"content": "cat_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269229": {
"content": "ceb_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269230": {
"content": "ces_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269231": {
"content": "cjk_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269232": {
"content": "ckb_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269233": {
"content": "crh_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269234": {
"content": "cym_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269235": {
"content": "dan_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269236": {
"content": "deu_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269237": {
"content": "dik_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269238": {
"content": "dyu_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269239": {
"content": "dzo_Tibt",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269240": {
"content": "ell_Grek",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269241": {
"content": "eng_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269242": {
"content": "epo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269243": {
"content": "est_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269244": {
"content": "eus_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269245": {
"content": "ewe_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269246": {
"content": "fao_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269247": {
"content": "pes_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269248": {
"content": "fij_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269249": {
"content": "fin_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269250": {
"content": "fon_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269251": {
"content": "fra_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269252": {
"content": "fur_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269253": {
"content": "fuv_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269254": {
"content": "gla_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269255": {
"content": "gle_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269256": {
"content": "glg_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269257": {
"content": "grn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269258": {
"content": "guj_Gujr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269259": {
"content": "hat_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269260": {
"content": "hau_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269261": {
"content": "heb_Hebr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269262": {
"content": "hin_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269263": {
"content": "hne_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269264": {
"content": "hrv_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269265": {
"content": "hun_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269266": {
"content": "hye_Armn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269267": {
"content": "ibo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269268": {
"content": "ilo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269269": {
"content": "ind_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269270": {
"content": "isl_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269271": {
"content": "ita_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269272": {
"content": "jav_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269273": {
"content": "jpn_Jpan",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269274": {
"content": "kab_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269275": {
"content": "kac_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269276": {
"content": "kam_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269277": {
"content": "kan_Knda",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269278": {
"content": "kas_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269279": {
"content": "kas_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269280": {
"content": "kat_Geor",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269281": {
"content": "knc_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269282": {
"content": "knc_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269283": {
"content": "kaz_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269284": {
"content": "kbp_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269285": {
"content": "kea_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269286": {
"content": "khm_Khmr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269287": {
"content": "kik_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269288": {
"content": "kin_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269289": {
"content": "kir_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269290": {
"content": "kmb_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269291": {
"content": "kon_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269292": {
"content": "kor_Hang",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269293": {
"content": "kmr_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269294": {
"content": "lao_Laoo",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269295": {
"content": "lvs_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269296": {
"content": "lij_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269297": {
"content": "lim_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269298": {
"content": "lin_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269299": {
"content": "lit_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269300": {
"content": "lmo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269301": {
"content": "ltg_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269302": {
"content": "ltz_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269303": {
"content": "lua_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269304": {
"content": "lug_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269305": {
"content": "luo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269306": {
"content": "lus_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269307": {
"content": "mag_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269308": {
"content": "mai_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269309": {
"content": "mal_Mlym",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269310": {
"content": "mar_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269311": {
"content": "min_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269312": {
"content": "mkd_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269313": {
"content": "plt_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269314": {
"content": "mlt_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269315": {
"content": "mni_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269316": {
"content": "khk_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269317": {
"content": "mos_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269318": {
"content": "mri_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269319": {
"content": "zsm_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269320": {
"content": "mya_Mymr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269321": {
"content": "nld_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269322": {
"content": "nno_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269323": {
"content": "nob_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269324": {
"content": "npi_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269325": {
"content": "nso_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269326": {
"content": "nus_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269327": {
"content": "nya_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269328": {
"content": "oci_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269329": {
"content": "gaz_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269330": {
"content": "ory_Orya",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269331": {
"content": "pag_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269332": {
"content": "pan_Guru",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269333": {
"content": "pap_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269334": {
"content": "pol_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269335": {
"content": "por_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269336": {
"content": "prs_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269337": {
"content": "pbt_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269338": {
"content": "quy_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269339": {
"content": "ron_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269340": {
"content": "run_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269341": {
"content": "rus_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269342": {
"content": "sag_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269343": {
"content": "san_Deva",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269344": {
"content": "sat_Beng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269345": {
"content": "scn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269346": {
"content": "shn_Mymr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269347": {
"content": "sin_Sinh",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269348": {
"content": "slk_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269349": {
"content": "slv_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269350": {
"content": "smo_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269351": {
"content": "sna_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269352": {
"content": "snd_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269353": {
"content": "som_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269354": {
"content": "sot_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269355": {
"content": "spa_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269356": {
"content": "als_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269357": {
"content": "srd_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269358": {
"content": "srp_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269359": {
"content": "ssw_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269360": {
"content": "sun_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269361": {
"content": "swe_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269362": {
"content": "swh_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269363": {
"content": "szl_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269364": {
"content": "tam_Taml",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269365": {
"content": "tat_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269366": {
"content": "tel_Telu",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269367": {
"content": "tgk_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269368": {
"content": "tgl_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269369": {
"content": "tha_Thai",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269370": {
"content": "tir_Ethi",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269371": {
"content": "taq_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269372": {
"content": "taq_Tfng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269373": {
"content": "tpi_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269374": {
"content": "tsn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269375": {
"content": "tso_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269376": {
"content": "tuk_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269377": {
"content": "tum_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269378": {
"content": "tur_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269379": {
"content": "twi_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269380": {
"content": "tzm_Tfng",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269381": {
"content": "uig_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269382": {
"content": "ukr_Cyrl",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269383": {
"content": "umb_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269384": {
"content": "urd_Arab",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269385": {
"content": "uzn_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269386": {
"content": "vec_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269387": {
"content": "vie_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269388": {
"content": "war_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269389": {
"content": "wol_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269390": {
"content": "xho_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269391": {
"content": "ydd_Hebr",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269392": {
"content": "yor_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269393": {
"content": "yue_Hant",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269394": {
"content": "zho_Hans",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269395": {
"content": "zho_Hant",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269396": {
"content": "zul_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"269397": {
"content": "<mask>",
"lstrip": true,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": true
},
"269398": {
"content": "kaa_Latn",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
}
},
"additional_special_tokens": [
"ace_Arab",
"ace_Latn",
"acm_Arab",
"acq_Arab",
"aeb_Arab",
"afr_Latn",
"ajp_Arab",
"aka_Latn",
"amh_Ethi",
"apc_Arab",
"arb_Arab",
"ars_Arab",
"ary_Arab",
"arz_Arab",
"asm_Beng",
"ast_Latn",
"awa_Deva",
"ayr_Latn",
"azb_Arab",
"azj_Latn",
"bak_Cyrl",
"bam_Latn",
"ban_Latn",
"bel_Cyrl",
"bem_Latn",
"ben_Beng",
"bho_Deva",
"bjn_Arab",
"bjn_Latn",
"bod_Tibt",
"bos_Latn",
"bug_Latn",
"bul_Cyrl",
"cat_Latn",
"ceb_Latn",
"ces_Latn",
"cjk_Latn",
"ckb_Arab",
"crh_Latn",
"cym_Latn",
"dan_Latn",
"deu_Latn",
"dik_Latn",
"dyu_Latn",
"dzo_Tibt",
"ell_Grek",
"eng_Latn",
"epo_Latn",
"est_Latn",
"eus_Latn",
"ewe_Latn",
"fao_Latn",
"pes_Arab",
"fij_Latn",
"fin_Latn",
"fon_Latn",
"fra_Latn",
"fur_Latn",
"fuv_Latn",
"gla_Latn",
"gle_Latn",
"glg_Latn",
"grn_Latn",
"guj_Gujr",
"hat_Latn",
"hau_Latn",
"heb_Hebr",
"hin_Deva",
"hne_Deva",
"hrv_Latn",
"hun_Latn",
"hye_Armn",
"ibo_Latn",
"ilo_Latn",
"ind_Latn",
"isl_Latn",
"ita_Latn",
"jav_Latn",
"jpn_Jpan",
"kab_Latn",
"kac_Latn",
"kam_Latn",
"kan_Knda",
"kas_Arab",
"kas_Deva",
"kat_Geor",
"knc_Arab",
"knc_Latn",
"kaz_Cyrl",
"kbp_Latn",
"kea_Latn",
"khm_Khmr",
"kik_Latn",
"kin_Latn",
"kir_Cyrl",
"kmb_Latn",
"kon_Latn",
"kor_Hang",
"kmr_Latn",
"lao_Laoo",
"lvs_Latn",
"lij_Latn",
"lim_Latn",
"lin_Latn",
"lit_Latn",
"lmo_Latn",
"ltg_Latn",
"ltz_Latn",
"lua_Latn",
"lug_Latn",
"luo_Latn",
"lus_Latn",
"mag_Deva",
"mai_Deva",
"mal_Mlym",
"mar_Deva",
"min_Latn",
"mkd_Cyrl",
"plt_Latn",
"mlt_Latn",
"mni_Beng",
"khk_Cyrl",
"mos_Latn",
"mri_Latn",
"zsm_Latn",
"mya_Mymr",
"nld_Latn",
"nno_Latn",
"nob_Latn",
"npi_Deva",
"nso_Latn",
"nus_Latn",
"nya_Latn",
"oci_Latn",
"gaz_Latn",
"ory_Orya",
"pag_Latn",
"pan_Guru",
"pap_Latn",
"pol_Latn",
"por_Latn",
"prs_Arab",
"pbt_Arab",
"quy_Latn",
"ron_Latn",
"run_Latn",
"rus_Cyrl",
"sag_Latn",
"san_Deva",
"sat_Beng",
"scn_Latn",
"shn_Mymr",
"sin_Sinh",
"slk_Latn",
"slv_Latn",
"smo_Latn",
"sna_Latn",
"snd_Arab",
"som_Latn",
"sot_Latn",
"spa_Latn",
"als_Latn",
"srd_Latn",
"srp_Cyrl",
"ssw_Latn",
"sun_Latn",
"swe_Latn",
"swh_Latn",
"szl_Latn",
"tam_Taml",
"tat_Cyrl",
"tel_Telu",
"tgk_Cyrl",
"tgl_Latn",
"tha_Thai",
"tir_Ethi",
"taq_Latn",
"taq_Tfng",
"tpi_Latn",
"tsn_Latn",
"tso_Latn",
"tuk_Latn",
"tum_Latn",
"tur_Latn",
"twi_Latn",
"tzm_Tfng",
"uig_Arab",
"ukr_Cyrl",
"umb_Latn",
"urd_Arab",
"uzn_Latn",
"vec_Latn",
"vie_Latn",
"war_Latn",
"wol_Latn",
"xho_Latn",
"ydd_Hebr",
"yor_Latn",
"yue_Hant",
"zho_Hans",
"zho_Hant",
"zul_Latn",
"kaa_Latn"
],
"bos_token": "<s>",
"clean_up_tokenization_spaces": true,
"cls_token": "<s>",
"eos_token": "</s>",
"legacy_behaviour": false,
"mask_token": "<mask>",
"model_max_length": 1024,
"pad_token": "<pad>",
"sep_token": "</s>",
"sp_model_kwargs": {},
"src_lang": "kaa_Latn",
"tgt_lang": "eng_Latn",
"tokenizer_class": "NllbTokenizer",
"unk_token": "<unk>"
}