joelb/mistral-7b-instruct-ccai-scotland
Browse files- README.md +2 -6
- adapter_config.json +1 -1
- adapter_model.safetensors +2 -2
- tokenizer_config.json +4 -0
- training_args.bin +1 -1
README.md
CHANGED
@@ -35,17 +35,13 @@ More information needed
|
|
35 |
### Training hyperparameters
|
36 |
|
37 |
The following hyperparameters were used during training:
|
38 |
-
- learning_rate:
|
39 |
- train_batch_size: 8
|
40 |
- eval_batch_size: 8
|
41 |
- seed: 42
|
42 |
- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08
|
43 |
- lr_scheduler_type: linear
|
44 |
-
- training_steps:
|
45 |
-
|
46 |
-
### Training results
|
47 |
-
|
48 |
-
|
49 |
|
50 |
### Framework versions
|
51 |
|
|
|
35 |
### Training hyperparameters
|
36 |
|
37 |
The following hyperparameters were used during training:
|
38 |
+
- learning_rate: 4e-05
|
39 |
- train_batch_size: 8
|
40 |
- eval_batch_size: 8
|
41 |
- seed: 42
|
42 |
- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08
|
43 |
- lr_scheduler_type: linear
|
44 |
+
- training_steps: 2500
|
|
|
|
|
|
|
|
|
45 |
|
46 |
### Framework versions
|
47 |
|
adapter_config.json
CHANGED
@@ -16,7 +16,7 @@
|
|
16 |
"megatron_core": "megatron.core",
|
17 |
"modules_to_save": null,
|
18 |
"peft_type": "LORA",
|
19 |
-
"r":
|
20 |
"rank_pattern": {},
|
21 |
"revision": null,
|
22 |
"target_modules": [
|
|
|
16 |
"megatron_core": "megatron.core",
|
17 |
"modules_to_save": null,
|
18 |
"peft_type": "LORA",
|
19 |
+
"r": 64,
|
20 |
"rank_pattern": {},
|
21 |
"revision": null,
|
22 |
"target_modules": [
|
adapter_model.safetensors
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
-
size
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:a570bc7c17a73dab9150e1228606feb00b120f6415f8fd4b05867884d2407925
|
3 |
+
size 109069176
|
tokenizer_config.json
CHANGED
@@ -33,11 +33,15 @@
|
|
33 |
"clean_up_tokenization_spaces": false,
|
34 |
"eos_token": "</s>",
|
35 |
"legacy": false,
|
|
|
36 |
"model_max_length": 1000000000000000019884624838656,
|
37 |
"pad_token": "</s>",
|
38 |
"sp_model_kwargs": {},
|
39 |
"spaces_between_special_tokens": false,
|
|
|
40 |
"tokenizer_class": "LlamaTokenizer",
|
|
|
|
|
41 |
"unk_token": "<unk>",
|
42 |
"use_default_system_prompt": false
|
43 |
}
|
|
|
33 |
"clean_up_tokenization_spaces": false,
|
34 |
"eos_token": "</s>",
|
35 |
"legacy": false,
|
36 |
+
"max_length": 1024,
|
37 |
"model_max_length": 1000000000000000019884624838656,
|
38 |
"pad_token": "</s>",
|
39 |
"sp_model_kwargs": {},
|
40 |
"spaces_between_special_tokens": false,
|
41 |
+
"stride": 0,
|
42 |
"tokenizer_class": "LlamaTokenizer",
|
43 |
+
"truncation_side": "right",
|
44 |
+
"truncation_strategy": "longest_first",
|
45 |
"unk_token": "<unk>",
|
46 |
"use_default_system_prompt": false
|
47 |
}
|
training_args.bin
CHANGED
@@ -1,3 +1,3 @@
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
-
oid sha256:
|
3 |
size 4984
|
|
|
1 |
version https://git-lfs.github.com/spec/v1
|
2 |
+
oid sha256:07dacbad5f929f75692c643f3fd2c20e49bfc7e66bac6419c5d0478e630be927
|
3 |
size 4984
|