rntc commited on
Commit
5b6abc9
1 Parent(s): df417e1

Upload tokenizer

Browse files
README.md CHANGED
@@ -1,19 +1,18 @@
1
  ---
2
- license: mit
3
  language:
4
  - fr
5
- pipeline_tag: fill-mask
 
6
  tags:
7
  - biomedical
8
  - clinical
9
  - life sciences
10
  datasets:
11
  - rntc/biomed-fr
 
12
  widget:
13
- - text: >-
14
- Les médicaments <mask> typiques sont largement utilisés dans le traitement
15
  de première intention des patients schizophrènes.
16
- library_name: transformers
17
  ---
18
 
19
  <a href=https://camembert-bio-model.fr/>
 
1
  ---
 
2
  language:
3
  - fr
4
+ license: mit
5
+ library_name: transformers
6
  tags:
7
  - biomedical
8
  - clinical
9
  - life sciences
10
  datasets:
11
  - rntc/biomed-fr
12
+ pipeline_tag: fill-mask
13
  widget:
14
+ - text: Les médicaments <mask> typiques sont largement utilisés dans le traitement
 
15
  de première intention des patients schizophrènes.
 
16
  ---
17
 
18
  <a href=https://camembert-bio-model.fr/>
added_tokens.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "<unk>NOTUSED": 32005
3
+ }
special_tokens_map.json CHANGED
@@ -1,7 +1,8 @@
1
  {
2
  "additional_special_tokens": [
3
  "<s>NOTUSED",
4
- "</s>NOTUSED"
 
5
  ],
6
  "bos_token": "<s>",
7
  "cls_token": "<s>",
 
1
  {
2
  "additional_special_tokens": [
3
  "<s>NOTUSED",
4
+ "</s>NOTUSED",
5
+ "<unk>NOTUSED"
6
  ],
7
  "bos_token": "<s>",
8
  "cls_token": "<s>",
tokenizer.json CHANGED
@@ -65,6 +65,15 @@
65
  "rstrip": false,
66
  "normalized": false,
67
  "special": true
 
 
 
 
 
 
 
 
 
68
  }
69
  ],
70
  "normalizer": {
@@ -80,7 +89,8 @@
80
  {
81
  "type": "Metaspace",
82
  "replacement": "▁",
83
- "add_prefix_space": true
 
84
  }
85
  ]
86
  },
@@ -168,7 +178,8 @@
168
  "decoder": {
169
  "type": "Metaspace",
170
  "replacement": "▁",
171
- "add_prefix_space": true
 
172
  },
173
  "model": {
174
  "type": "Unigram",
@@ -128194,6 +128205,7 @@
128194
  "<mask>",
128195
  0.0
128196
  ]
128197
- ]
 
128198
  }
128199
  }
 
65
  "rstrip": false,
66
  "normalized": false,
67
  "special": true
68
+ },
69
+ {
70
+ "id": 32005,
71
+ "content": "<unk>NOTUSED",
72
+ "single_word": false,
73
+ "lstrip": false,
74
+ "rstrip": false,
75
+ "normalized": false,
76
+ "special": true
77
  }
78
  ],
79
  "normalizer": {
 
89
  {
90
  "type": "Metaspace",
91
  "replacement": "▁",
92
+ "prepend_scheme": "always",
93
+ "split": true
94
  }
95
  ]
96
  },
 
178
  "decoder": {
179
  "type": "Metaspace",
180
  "replacement": "▁",
181
+ "prepend_scheme": "always",
182
+ "split": true
183
  },
184
  "model": {
185
  "type": "Unigram",
 
128205
  "<mask>",
128206
  0.0
128207
  ]
128208
+ ],
128209
+ "byte_fallback": false
128210
  }
128211
  }
tokenizer_config.json CHANGED
@@ -1,24 +1,83 @@
1
  {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2
  "additional_special_tokens": [
3
  "<s>NOTUSED",
4
- "</s>NOTUSED"
 
5
  ],
6
  "bos_token": "<s>",
 
7
  "cls_token": "<s>",
8
  "eos_token": "</s>",
9
- "mask_token": {
10
- "__type": "AddedToken",
11
- "content": "<mask>",
12
- "lstrip": true,
13
- "normalized": true,
14
- "rstrip": false,
15
- "single_word": false
16
- },
17
  "model_max_length": 512,
18
- "name_or_path": "camembert-base",
19
  "pad_token": "<pad>",
20
  "sep_token": "</s>",
21
- "special_tokens_map_file": null,
22
  "tokenizer_class": "CamembertTokenizer",
23
  "unk_token": "<unk>"
24
  }
 
1
  {
2
+ "added_tokens_decoder": {
3
+ "0": {
4
+ "content": "<s>NOTUSED",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ },
11
+ "1": {
12
+ "content": "<pad>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false,
17
+ "special": true
18
+ },
19
+ "2": {
20
+ "content": "</s>NOTUSED",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false,
25
+ "special": true
26
+ },
27
+ "4": {
28
+ "content": "<unk>",
29
+ "lstrip": false,
30
+ "normalized": false,
31
+ "rstrip": false,
32
+ "single_word": false,
33
+ "special": true
34
+ },
35
+ "5": {
36
+ "content": "<s>",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": false,
40
+ "single_word": false,
41
+ "special": true
42
+ },
43
+ "6": {
44
+ "content": "</s>",
45
+ "lstrip": false,
46
+ "normalized": false,
47
+ "rstrip": false,
48
+ "single_word": false,
49
+ "special": true
50
+ },
51
+ "32004": {
52
+ "content": "<mask>",
53
+ "lstrip": true,
54
+ "normalized": false,
55
+ "rstrip": false,
56
+ "single_word": false,
57
+ "special": true
58
+ },
59
+ "32005": {
60
+ "content": "<unk>NOTUSED",
61
+ "lstrip": false,
62
+ "normalized": false,
63
+ "rstrip": false,
64
+ "single_word": false,
65
+ "special": true
66
+ }
67
+ },
68
  "additional_special_tokens": [
69
  "<s>NOTUSED",
70
+ "</s>NOTUSED",
71
+ "<unk>NOTUSED"
72
  ],
73
  "bos_token": "<s>",
74
+ "clean_up_tokenization_spaces": true,
75
  "cls_token": "<s>",
76
  "eos_token": "</s>",
77
+ "mask_token": "<mask>",
 
 
 
 
 
 
 
78
  "model_max_length": 512,
 
79
  "pad_token": "<pad>",
80
  "sep_token": "</s>",
 
81
  "tokenizer_class": "CamembertTokenizer",
82
  "unk_token": "<unk>"
83
  }