rntc commited on
Commit
7f8fe26
1 Parent(s): 5b6abc9

revert tokenizer

Browse files
README.md CHANGED
@@ -1,18 +1,19 @@
1
  ---
 
2
  language:
3
  - fr
4
- license: mit
5
- library_name: transformers
6
  tags:
7
  - biomedical
8
  - clinical
9
  - life sciences
10
  datasets:
11
  - rntc/biomed-fr
12
- pipeline_tag: fill-mask
13
  widget:
14
- - text: Les médicaments <mask> typiques sont largement utilisés dans le traitement
 
15
  de première intention des patients schizophrènes.
 
16
  ---
17
 
18
  <a href=https://camembert-bio-model.fr/>
 
1
  ---
2
+ license: mit
3
  language:
4
  - fr
5
+ pipeline_tag: fill-mask
 
6
  tags:
7
  - biomedical
8
  - clinical
9
  - life sciences
10
  datasets:
11
  - rntc/biomed-fr
 
12
  widget:
13
+ - text: >-
14
+ Les médicaments <mask> typiques sont largement utilisés dans le traitement
15
  de première intention des patients schizophrènes.
16
+ library_name: transformers
17
  ---
18
 
19
  <a href=https://camembert-bio-model.fr/>
added_tokens.json DELETED
@@ -1,3 +0,0 @@
1
- {
2
- "<unk>NOTUSED": 32005
3
- }
 
 
 
 
special_tokens_map.json CHANGED
@@ -1,8 +1,7 @@
1
  {
2
  "additional_special_tokens": [
3
  "<s>NOTUSED",
4
- "</s>NOTUSED",
5
- "<unk>NOTUSED"
6
  ],
7
  "bos_token": "<s>",
8
  "cls_token": "<s>",
 
1
  {
2
  "additional_special_tokens": [
3
  "<s>NOTUSED",
4
+ "</s>NOTUSED"
 
5
  ],
6
  "bos_token": "<s>",
7
  "cls_token": "<s>",
tokenizer.json CHANGED
@@ -65,15 +65,6 @@
65
  "rstrip": false,
66
  "normalized": false,
67
  "special": true
68
- },
69
- {
70
- "id": 32005,
71
- "content": "<unk>NOTUSED",
72
- "single_word": false,
73
- "lstrip": false,
74
- "rstrip": false,
75
- "normalized": false,
76
- "special": true
77
  }
78
  ],
79
  "normalizer": {
@@ -89,8 +80,7 @@
89
  {
90
  "type": "Metaspace",
91
  "replacement": "▁",
92
- "prepend_scheme": "always",
93
- "split": true
94
  }
95
  ]
96
  },
@@ -178,8 +168,7 @@
178
  "decoder": {
179
  "type": "Metaspace",
180
  "replacement": "▁",
181
- "prepend_scheme": "always",
182
- "split": true
183
  },
184
  "model": {
185
  "type": "Unigram",
@@ -128205,7 +128194,6 @@
128205
  "<mask>",
128206
  0.0
128207
  ]
128208
- ],
128209
- "byte_fallback": false
128210
  }
128211
  }
 
65
  "rstrip": false,
66
  "normalized": false,
67
  "special": true
 
 
 
 
 
 
 
 
 
68
  }
69
  ],
70
  "normalizer": {
 
80
  {
81
  "type": "Metaspace",
82
  "replacement": "▁",
83
+ "add_prefix_space": true
 
84
  }
85
  ]
86
  },
 
168
  "decoder": {
169
  "type": "Metaspace",
170
  "replacement": "▁",
171
+ "add_prefix_space": true
 
172
  },
173
  "model": {
174
  "type": "Unigram",
 
128194
  "<mask>",
128195
  0.0
128196
  ]
128197
+ ]
 
128198
  }
128199
  }
tokenizer_config.json CHANGED
@@ -1,83 +1,24 @@
1
  {
2
- "added_tokens_decoder": {
3
- "0": {
4
- "content": "<s>NOTUSED",
5
- "lstrip": false,
6
- "normalized": false,
7
- "rstrip": false,
8
- "single_word": false,
9
- "special": true
10
- },
11
- "1": {
12
- "content": "<pad>",
13
- "lstrip": false,
14
- "normalized": false,
15
- "rstrip": false,
16
- "single_word": false,
17
- "special": true
18
- },
19
- "2": {
20
- "content": "</s>NOTUSED",
21
- "lstrip": false,
22
- "normalized": false,
23
- "rstrip": false,
24
- "single_word": false,
25
- "special": true
26
- },
27
- "4": {
28
- "content": "<unk>",
29
- "lstrip": false,
30
- "normalized": false,
31
- "rstrip": false,
32
- "single_word": false,
33
- "special": true
34
- },
35
- "5": {
36
- "content": "<s>",
37
- "lstrip": false,
38
- "normalized": false,
39
- "rstrip": false,
40
- "single_word": false,
41
- "special": true
42
- },
43
- "6": {
44
- "content": "</s>",
45
- "lstrip": false,
46
- "normalized": false,
47
- "rstrip": false,
48
- "single_word": false,
49
- "special": true
50
- },
51
- "32004": {
52
- "content": "<mask>",
53
- "lstrip": true,
54
- "normalized": false,
55
- "rstrip": false,
56
- "single_word": false,
57
- "special": true
58
- },
59
- "32005": {
60
- "content": "<unk>NOTUSED",
61
- "lstrip": false,
62
- "normalized": false,
63
- "rstrip": false,
64
- "single_word": false,
65
- "special": true
66
- }
67
- },
68
  "additional_special_tokens": [
69
  "<s>NOTUSED",
70
- "</s>NOTUSED",
71
- "<unk>NOTUSED"
72
  ],
73
  "bos_token": "<s>",
74
- "clean_up_tokenization_spaces": true,
75
  "cls_token": "<s>",
76
  "eos_token": "</s>",
77
- "mask_token": "<mask>",
 
 
 
 
 
 
 
78
  "model_max_length": 512,
 
79
  "pad_token": "<pad>",
80
  "sep_token": "</s>",
 
81
  "tokenizer_class": "CamembertTokenizer",
82
  "unk_token": "<unk>"
83
  }
 
1
  {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2
  "additional_special_tokens": [
3
  "<s>NOTUSED",
4
+ "</s>NOTUSED"
 
5
  ],
6
  "bos_token": "<s>",
 
7
  "cls_token": "<s>",
8
  "eos_token": "</s>",
9
+ "mask_token": {
10
+ "__type": "AddedToken",
11
+ "content": "<mask>",
12
+ "lstrip": true,
13
+ "normalized": true,
14
+ "rstrip": false,
15
+ "single_word": false
16
+ },
17
  "model_max_length": 512,
18
+ "name_or_path": "camembert-base",
19
  "pad_token": "<pad>",
20
  "sep_token": "</s>",
21
+ "special_tokens_map_file": null,
22
  "tokenizer_class": "CamembertTokenizer",
23
  "unk_token": "<unk>"
24
  }