{ "version": "1.0", "truncation": null, "padding": null, "added_tokens": [ { "id": 0, "content": "[UNK]", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 1, "content": "[CLS]", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 2, "content": "[SEP]", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 3, "content": "[PAD]", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true }, { "id": 4, "content": "[MASK]", "single_word": false, "lstrip": false, "rstrip": false, "normalized": false, "special": true } ], "normalizer": null, "pre_tokenizer": { "type": "Whitespace" }, "post_processor": null, "decoder": null, "model": { "type": "BPE", "dropout": null, "unk_token": "[UNK]", "continuing_subword_prefix": null, "end_of_word_suffix": null, "fuse_unk": false, "byte_fallback": false, "vocab": { "[UNK]": 0, "[CLS]": 1, "[SEP]": 2, "[PAD]": 3, "[MASK]": 4, "!": 5, "#": 6, "*": 7, ",": 8, "-": 9, ".": 10, "=": 11, "?": 12, "N": 13, "Q": 14, "^": 15, "_": 16, "`": 17, "a": 18, "b": 19, "d": 20, "e": 21, "f": 22, "g": 23, "h": 24, "i": 25, "j": 26, "k": 27, "l": 28, "m": 29, "n": 30, "o": 31, "p": 32, "s": 33, "t": 34, "u": 35, "v": 36, "w": 37, "x": 38, "y": 39, "z": 40, "~": 41, "æ": 42, "ç": 43, "ð": 44, "ŋ": 45, "ɑ": 46, "ɔ": 47, "ə": 48, "ɛ": 49, "ɥ": 50, "ɪ": 51, "ɫ": 52, "ɯ": 53, "ɸ": 54, "ɹ": 55, "ɾ": 56, "ʃ": 57, "ʊ": 58, "ʑ": 59, "ʒ": 60, "ʰ": 61, "ˈ": 62, "ˌ": 63, "θ": 64, "…": 65, "⁼": 66, "↑": 67, "→": 68, "↓": 69, "ɯəj": 70, "ɤ̆j": 71, "ʷiə": 72, "ɤ̆w": 73, "ɯəw": 74, "ʷet": 75, "iəw": 76, "uəj": 77, "ʷen": 78, "tʰw": 79, "ʷɤ̆": 80, "ʷiu": 81, "kwi": 82, "ŋ͡m": 83, "k͡p": 84, "cw": 85, "jw": 86, "uə": 87, "eə": 88, "bw": 89, "oj": 90, "ʷi": 91, "vw": 92, "ăw": 93, "ʈw": 94, "ʂw": 95, "aʊ": 96, "fw": 97, "ɛu": 98, "tʰ": 99, "tʃ": 100, "ɔɪ": 101, "xw": 102, "ʷɤ": 103, "ɤ̆": 104, "ŋw": 105, "ʊə": 106, "zi": 107, "ʷă": 108, "dw": 109, "eɪ": 110, "aɪ": 111, "ew": 112, "iə": 113, "ɣw": 114, "zw": 115, "ɯj": 116, "ʷɛ": 117, "ɯw": 118, "ɤj": 119, "ɔ:": 120, "əʊ": 121, "ʷa": 122, "mw": 123, "ɑ:": 124, "hw": 125, "ɔj": 126, "uj": 127, "lw": 128, "ɪə": 129, "ăj": 130, "u:": 131, "aw": 132, "ɛj": 133, "iw": 134, "aj": 135, "ɜ:": 136, "kw": 137, "nw": 138, "t∫": 139, "ɲw": 140, "eo": 141, "sw": 142, "tw": 143, "ʐw": 144, "iɛ": 145, "ʷe": 146, "i:": 147, "ɯə": 148, "dʒ": 149, "ɲ": 150, "ʌ": 151, "1": 152, "∫": 153, "3": 154, "ɣ": 155, "ʧ": 156, "6": 157, "ʐ": 158, "ă": 159, "ɤ": 160, "2": 161, "ʤ": 162, "ɒ": 163, "ʂ": 164, "5": 165, " ": 166, "c": 167, "ʈ": 168, "4": 169, "r": 170, ":": 171, "η": 172, ";": 173, "'": 174 }, "merges": [ ] } }