lapp0 commited on
Commit
e7564aa
1 Parent(s): c9060ec

End of training

Browse files
README.md CHANGED
@@ -17,13 +17,13 @@ The [Distily](https://github.com/lapp0/distily) library was used for this distil
17
 
18
  It achieves the following results on the evaluation set:
19
  - eval_enwikippl: 84.0
20
- - eval_frwikippl: 342.0
21
- - eval_zhwikippl: 217.0
22
- - eval_tinystoriesppl: 69.5
23
- - eval_loss: 0.6877
24
- - eval_runtime: 16.9969
25
- - eval_samples_per_second: 58.834
26
- - eval_steps_per_second: 7.354
27
 
28
  <!-- This model card has been generated automatically according to the information the Trainer had access to. You
29
  should probably proofread and complete it, then remove this comment.
@@ -64,32 +64,32 @@ Peak GPU Memory: 7.7252 GB
64
  | step | epoch | enwikippl | frwikippl | loss | runtime | samples_per_second | steps_per_second | tinystoriesppl | zhwikippl |
65
  | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
66
  | **teacher eval** | | 43.75 | 61.75 | | | | | 11.8125 | 19.125 |
67
- | 0 | 0 | 2473901162496.0 | 170424302305280.0 | 20.7680 | 17.0409 | 58.682 | 7.335 | 4060086272.0 | 71468255805440.0 |
68
- | 1000 | 0.0404 | 334.0 | 1464.0 | 1.5419 | 17.0178 | 58.762 | 7.345 | 243.0 | 596.0 |
69
- | 2000 | 0.0808 | 232.0 | 756.0 | 1.3235 | 16.9755 | 58.909 | 7.364 | 189.0 | 250.0 |
70
- | 3000 | 0.1212 | 180.0 | 628.0 | 1.1620 | 16.9923 | 58.85 | 7.356 | 149.0 | 171.0 |
71
- | 4000 | 0.1616 | 150.0 | 576.0 | 1.0434 | 16.9803 | 58.892 | 7.361 | 121.5 | 172.0 |
72
- | 5000 | 0.2020 | 130.0 | 504.0 | 0.9520 | 17.0128 | 58.779 | 7.347 | 100.5 | 144.0 |
73
- | 6000 | 0.2424 | 113.5 | 420.0 | 0.8702 | 17.0074 | 58.798 | 7.35 | 91.0 | 137.0 |
74
- | 7000 | 0.2828 | 106.0 | 408.0 | 0.8100 | 16.9821 | 58.885 | 7.361 | 80.5 | 160.0 |
75
- | 8000 | 0.3232 | 96.5 | 396.0 | 0.7421 | 16.9749 | 58.911 | 7.364 | 70.5 | 127.0 |
76
- | 9000 | 0.3636 | 84.0 | 342.0 | 0.6877 | 16.9969 | 58.834 | 7.354 | 69.5 | 217.0 |
77
- | 10000 | 0.4040 | 78.0 | 300.0 | 0.6467 | 16.9846 | 58.877 | 7.36 | 65.0 | 139.0 |
78
- | 11000 | 0.4444 | 77.0 | 278.0 | 0.5957 | 16.9903 | 58.857 | 7.357 | 60.0 | 127.5 |
79
- | 12000 | 0.4848 | 75.0 | 272.0 | 0.5789 | 16.9858 | 58.873 | 7.359 | 56.5 | 140.0 |
80
- | 13000 | 0.5253 | 71.5 | 266.0 | 0.5525 | 16.9418 | 59.026 | 7.378 | 56.5 | 116.0 |
81
- | 14000 | 0.5657 | 71.0 | 252.0 | 0.5416 | 17.088 | 58.521 | 7.315 | 53.75 | 132.0 |
82
- | 15000 | 0.6061 | 68.0 | 221.0 | 0.5283 | 16.9524 | 58.989 | 7.374 | 50.25 | 112.5 |
83
- | 16000 | 0.6465 | 70.0 | 244.0 | 0.5200 | 17.0495 | 58.653 | 7.332 | 52.5 | 109.5 |
84
- | 17000 | 0.6869 | 67.0 | 225.0 | 0.5097 | 17.0223 | 58.747 | 7.343 | 51.5 | 109.0 |
85
- | 18000 | 0.7273 | 71.0 | 239.0 | 0.5016 | 17.0519 | 58.644 | 7.331 | 49.5 | 150.0 |
86
- | 19000 | 0.7677 | 68.0 | 212.0 | 0.4887 | 17.0831 | 58.537 | 7.317 | 51.25 | 98.0 |
87
- | 20000 | 0.8081 | 65.0 | 211.0 | 0.4865 | 17.0098 | 58.789 | 7.349 | 49.0 | 101.5 |
88
- | 21000 | 0.8485 | 64.5 | 217.0 | 0.4791 | 17.0253 | 58.736 | 7.342 | 47.5 | 142.0 |
89
- | 22000 | 0.8889 | 66.5 | 230.0 | 0.4798 | 16.9954 | 58.839 | 7.355 | 48.5 | 147.0 |
90
- | 23000 | 0.9293 | 62.5 | 212.0 | 0.4675 | 16.9835 | 58.881 | 7.36 | 45.5 | 134.0 |
91
- | 24000 | 0.9697 | 63.5 | 220.0 | 0.4712 | 16.9973 | 58.833 | 7.354 | 47.0 | 138.0 |
92
- | 24750 | 1.0 | 63.75 | 247.0 | 0.4679 | 17.0597 | 58.618 | 7.327 | 45.75 | 205.0 |
93
 
94
  ### Framework versions
95
  - Distily 0.2.0
 
17
 
18
  It achieves the following results on the evaluation set:
19
  - eval_enwikippl: 84.0
20
+ - eval_frwikippl: 336.0
21
+ - eval_zhwikippl: 143.0
22
+ - eval_tinystoriesppl: 68.0
23
+ - eval_loss: 0.6821
24
+ - eval_runtime: 16.9876
25
+ - eval_samples_per_second: 58.866
26
+ - eval_steps_per_second: 7.358
27
 
28
  <!-- This model card has been generated automatically according to the information the Trainer had access to. You
29
  should probably proofread and complete it, then remove this comment.
 
64
  | step | epoch | enwikippl | frwikippl | loss | runtime | samples_per_second | steps_per_second | tinystoriesppl | zhwikippl |
65
  | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
66
  | **teacher eval** | | 43.75 | 61.75 | | | | | 11.8125 | 19.125 |
67
+ | 0 | 0 | 2473901162496.0 | 170424302305280.0 | 20.7680 | 16.9731 | 58.917 | 7.365 | 4060086272.0 | 71468255805440.0 |
68
+ | 1000 | 0.0404 | 328.0 | 1488.0 | 1.5338 | 17.0565 | 58.629 | 7.329 | 243.0 | 608.0 |
69
+ | 2000 | 0.0808 | 229.0 | 792.0 | 1.3211 | 16.9911 | 58.854 | 7.357 | 190.0 | 260.0 |
70
+ | 3000 | 0.1212 | 178.0 | 624.0 | 1.1600 | 17.0018 | 58.817 | 7.352 | 149.0 | 176.0 |
71
+ | 4000 | 0.1616 | 147.0 | 580.0 | 1.0397 | 17.0215 | 58.749 | 7.344 | 119.0 | 161.0 |
72
+ | 5000 | 0.2020 | 128.0 | 516.0 | 0.9532 | 17.0153 | 58.771 | 7.346 | 102.0 | 159.0 |
73
+ | 6000 | 0.2424 | 111.0 | 410.0 | 0.8655 | 17.0046 | 58.808 | 7.351 | 90.0 | 147.0 |
74
+ | 7000 | 0.2828 | 104.5 | 410.0 | 0.8083 | 16.9742 | 58.913 | 7.364 | 82.0 | 145.0 |
75
+ | 8000 | 0.3232 | 97.5 | 382.0 | 0.7412 | 16.9735 | 58.915 | 7.364 | 74.0 | 128.0 |
76
+ | 9000 | 0.3636 | 84.0 | 336.0 | 0.6821 | 16.9876 | 58.866 | 7.358 | 68.0 | 143.0 |
77
+ | 10000 | 0.4040 | 77.5 | 312.0 | 0.6396 | 16.9771 | 58.903 | 7.363 | 65.0 | 140.0 |
78
+ | 11000 | 0.4444 | 75.5 | 280.0 | 0.5964 | 17.02 | 58.754 | 7.344 | 60.75 | 122.5 |
79
+ | 12000 | 0.4848 | 74.5 | 268.0 | 0.5797 | 16.9985 | 58.829 | 7.354 | 58.0 | 152.0 |
80
+ | 13000 | 0.5253 | 71.5 | 274.0 | 0.5537 | 16.9566 | 58.974 | 7.372 | 58.25 | 134.0 |
81
+ | 14000 | 0.5657 | 72.0 | 252.0 | 0.5429 | 16.9325 | 59.058 | 7.382 | 58.0 | 99.0 |
82
+ | 15000 | 0.6061 | 69.0 | 229.0 | 0.5308 | 16.9917 | 58.852 | 7.357 | 51.25 | 94.0 |
83
+ | 16000 | 0.6465 | 67.0 | 223.0 | 0.5209 | 16.9686 | 58.932 | 7.367 | 52.5 | 108.0 |
84
+ | 17000 | 0.6869 | 67.5 | 227.0 | 0.5046 | 16.979 | 58.896 | 7.362 | 54.25 | 118.0 |
85
+ | 18000 | 0.7273 | 67.5 | 244.0 | 0.5024 | 16.994 | 58.844 | 7.356 | 50.5 | 128.0 |
86
+ | 19000 | 0.7677 | 66.0 | 212.0 | 0.4931 | 16.9719 | 58.921 | 7.365 | 49.25 | 88.0 |
87
+ | 20000 | 0.8081 | 64.5 | 202.0 | 0.4925 | 17.0171 | 58.764 | 7.346 | 49.75 | 169.0 |
88
+ | 21000 | 0.8485 | 67.0 | 222.0 | 0.4839 | 16.9754 | 58.909 | 7.364 | 47.75 | 126.0 |
89
+ | 22000 | 0.8889 | 66.0 | 227.0 | 0.4759 | 16.9314 | 59.062 | 7.383 | 48.0 | 100.0 |
90
+ | 23000 | 0.9293 | 61.75 | 208.0 | 0.4704 | 16.9662 | 58.941 | 7.368 | 47.25 | 125.5 |
91
+ | 24000 | 0.9697 | 66.0 | 210.0 | 0.4706 | 17.0394 | 58.688 | 7.336 | 47.5 | 173.0 |
92
+ | 24750 | 1.0 | 63.75 | 218.0 | 0.4686 | 16.9798 | 58.894 | 7.362 | 46.75 | 82.5 |
93
 
94
  ### Framework versions
95
  - Distily 0.2.0
logs/hs_layer_mapper=last, hs_loss_fn=mse, hs_weight=1.0, learning_rate=0.0001, per_device_train_batch_size=4/events.out.tfevents.1724162747.02dbb11e2dcc ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c558a88ef09bfeed76cbfdbd2d86f1139add28429cd704b08c3514d25d302836
3
+ size 312