AdaptiveLayerLoss(model=model,
Browse filesloss=train_loss,
n_layers_per_step = 1,
last_layer_weight = 1,
prior_layers_weight= 1,
kl_div_weight = 1,
kl_temperature= 1,
)''')
lr = 1e-6. batch = 42, schedule = cosine