|
{ |
|
"_name_or_path": "yairschiff/caduceus_base", |
|
"architectures": [ |
|
"Caduceus" |
|
], |
|
"auto_map": { |
|
"AutoConfig": "yairschiff/caduceus_base--configuration_caduceus.CaduceusConfig", |
|
"AutoModel": "yairschiff/caduceus_base--modeling_caduceus.Caduceus", |
|
"AutoModelForMaskedLM": "yairschiff/caduceus_base--modeling_caduceus.CaduceusForMaskedLM", |
|
"AutoModelForSequenceClassification": "yairschiff/caduceus_base--modeling_caduceus.CaduceusForSequenceClassification" |
|
}, |
|
"bidirectional": true, |
|
"bidirectional_strategy": "add", |
|
"bidirectional_weight_tie": true, |
|
"complement_map": { |
|
"0": 0, |
|
"1": 1, |
|
"2": 2, |
|
"3": 6, |
|
"4": 5, |
|
"5": 4, |
|
"6": 3, |
|
"7": 7 |
|
}, |
|
"d_model": 512, |
|
"fused_add_norm": true, |
|
"initializer_cfg": { |
|
"initializer_range": 0.02, |
|
"n_residuals_per_layer": 1, |
|
"rescale_prenorm_residual": true |
|
}, |
|
"model_type": "caduceus", |
|
"n_layer": 24, |
|
"norm_epsilon": 1e-05, |
|
"pad_token_id": -100, |
|
"pad_vocab_size_multiple": 8, |
|
"rcps": true, |
|
"residual_in_fp32": true, |
|
"rms_norm": true, |
|
"ssm_cfg": { |
|
"bias": false, |
|
"conv_bias": true, |
|
"d_conv": 4, |
|
"d_state": 16, |
|
"dt_init": "random", |
|
"dt_init_floor": 0.0001, |
|
"dt_max": 0.1, |
|
"dt_min": 0.001, |
|
"dt_rank": "auto", |
|
"dt_scale": 1.0, |
|
"expand": 2, |
|
"use_fast_path": true |
|
}, |
|
"torch_dtype": "float32", |
|
"transformers_version": "4.37.2", |
|
"vocab_size": 8 |
|
} |
|
|