jonik commited on
Commit
df1834a
1 Parent(s): ddc568c

Upload tokenizer

Browse files
Files changed (2) hide show
  1. tokenizer_config.json +1 -0
  2. vocab.json +26 -26
tokenizer_config.json CHANGED
@@ -6,6 +6,7 @@
6
  "model_max_length": 1000000000000000019884624838656,
7
  "pad_token": "[PAD]",
8
  "replace_word_delimiter_char": " ",
 
9
  "tokenizer_class": "Wav2Vec2CTCTokenizer",
10
  "unk_token": "[UNK]",
11
  "word_delimiter_token": "|"
 
6
  "model_max_length": 1000000000000000019884624838656,
7
  "pad_token": "[PAD]",
8
  "replace_word_delimiter_char": " ",
9
+ "target_lang": null,
10
  "tokenizer_class": "Wav2Vec2CTCTokenizer",
11
  "unk_token": "[UNK]",
12
  "word_delimiter_token": "|"
vocab.json CHANGED
@@ -1,32 +1,32 @@
1
  {
2
- "'": 23,
3
- "A": 15,
4
- "B": 16,
5
- "C": 24,
6
- "D": 19,
7
- "E": 25,
8
- "F": 0,
9
- "G": 20,
10
- "H": 14,
11
- "I": 1,
12
  "J": 21,
13
- "K": 3,
14
  "L": 13,
15
- "M": 27,
16
- "N": 17,
17
- "O": 9,
18
- "P": 18,
19
- "Q": 11,
20
- "R": 22,
21
- "S": 10,
22
- "T": 5,
23
- "U": 6,
24
- "V": 2,
25
- "W": 8,
26
- "X": 26,
27
- "Y": 12,
28
- "Z": 4,
29
  "[PAD]": 29,
30
  "[UNK]": 28,
31
- "|": 7
32
  }
 
1
  {
2
+ "'": 5,
3
+ "A": 9,
4
+ "B": 25,
5
+ "C": 23,
6
+ "D": 20,
7
+ "E": 2,
8
+ "F": 19,
9
+ "G": 24,
10
+ "H": 18,
11
+ "I": 27,
12
  "J": 21,
13
+ "K": 22,
14
  "L": 13,
15
+ "M": 10,
16
+ "N": 0,
17
+ "O": 11,
18
+ "P": 17,
19
+ "Q": 14,
20
+ "R": 1,
21
+ "S": 26,
22
+ "T": 3,
23
+ "U": 8,
24
+ "V": 15,
25
+ "W": 6,
26
+ "X": 7,
27
+ "Y": 4,
28
+ "Z": 16,
29
  "[PAD]": 29,
30
  "[UNK]": 28,
31
+ "|": 12
32
  }