Fotini98 commited on
Commit
f6450e3
·
verified ·
1 Parent(s): 9188aac

Upload tokenizer

Browse files
Files changed (3) hide show
  1. added_tokens.json +2 -1
  2. tokenizer_config.json +0 -1
  3. vocab.json +28 -28
added_tokens.json CHANGED
@@ -1,4 +1,5 @@
1
  {
2
  "</s>": 31,
3
- "<s>": 30
 
4
  }
 
1
  {
2
  "</s>": 31,
3
+ "<s>": 30,
4
+ "[PAD]": 29
5
  }
tokenizer_config.json CHANGED
@@ -39,7 +39,6 @@
39
  "eos_token": "</s>",
40
  "model_max_length": 1000000000000000019884624838656,
41
  "pad_token": "[PAD]",
42
- "processor_class": "Wav2Vec2BertProcessor",
43
  "replace_word_delimiter_char": " ",
44
  "target_lang": null,
45
  "tokenizer_class": "Wav2Vec2CTCTokenizer",
 
39
  "eos_token": "</s>",
40
  "model_max_length": 1000000000000000019884624838656,
41
  "pad_token": "[PAD]",
 
42
  "replace_word_delimiter_char": " ",
43
  "target_lang": null,
44
  "tokenizer_class": "Wav2Vec2CTCTokenizer",
vocab.json CHANGED
@@ -1,32 +1,32 @@
1
  {
2
- "'": 15,
3
- "A": 10,
4
- "B": 24,
5
- "C": 11,
6
- "D": 26,
7
- "E": 25,
8
- "F": 2,
9
- "G": 27,
10
- "H": 0,
11
- "I": 23,
12
- "J": 13,
13
- "K": 5,
14
- "L": 19,
15
- "M": 9,
16
- "N": 8,
17
- "O": 12,
18
- "P": 17,
19
- "Q": 1,
20
- "R": 7,
21
- "S": 20,
22
- "T": 14,
23
- "U": 22,
24
- "V": 21,
25
- "W": 18,
26
- "X": 4,
27
- "Y": 6,
28
- "Z": 3,
29
  "[PAD]": 29,
30
  "[UNK]": 28,
31
- "|": 16
32
  }
 
1
  {
2
+ "'": 26,
3
+ "A": 7,
4
+ "B": 5,
5
+ "C": 3,
6
+ "D": 24,
7
+ "E": 10,
8
+ "F": 15,
9
+ "G": 19,
10
+ "H": 16,
11
+ "I": 8,
12
+ "J": 27,
13
+ "K": 25,
14
+ "L": 9,
15
+ "M": 22,
16
+ "N": 12,
17
+ "O": 11,
18
+ "P": 18,
19
+ "Q": 28,
20
+ "R": 2,
21
+ "S": 17,
22
+ "T": 6,
23
+ "U": 20,
24
+ "V": 13,
25
+ "W": 4,
26
+ "X": 21,
27
+ "Y": 23,
28
+ "Z": 14,
29
  "[PAD]": 29,
30
  "[UNK]": 28,
31
+ "|": 1
32
  }