Necent's picture
Upload folder using huggingface_hub
dd45c80 verified
Raw
History Blame Contribute Delete
5.65 kB
{
"best_global_step": null,
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 1.0,
"eval_steps": 500,
"global_step": 1423,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.035137034434293744,
"grad_norm": 13.614160537719727,
"learning_rate": 2.4500000000000003e-06,
"loss": 0.6868,
"step": 50
},
{
"epoch": 0.07027406886858749,
"grad_norm": 7.1465582847595215,
"learning_rate": 4.95e-06,
"loss": 0.5593,
"step": 100
},
{
"epoch": 0.10541110330288124,
"grad_norm": 0.4900398254394531,
"learning_rate": 7.450000000000001e-06,
"loss": 0.545,
"step": 150
},
{
"epoch": 0.14054813773717498,
"grad_norm": 0.034149572253227234,
"learning_rate": 9.950000000000001e-06,
"loss": 0.2203,
"step": 200
},
{
"epoch": 0.17568517217146873,
"grad_norm": 2.6971633434295654,
"learning_rate": 9.814814814814815e-06,
"loss": 0.1222,
"step": 250
},
{
"epoch": 0.21082220660576248,
"grad_norm": 0.012744505889713764,
"learning_rate": 9.625850340136055e-06,
"loss": 0.0908,
"step": 300
},
{
"epoch": 0.24595924104005623,
"grad_norm": 0.008385892026126385,
"learning_rate": 9.436885865457295e-06,
"loss": 0.0744,
"step": 350
},
{
"epoch": 0.28109627547434995,
"grad_norm": 3.446326494216919,
"learning_rate": 9.247921390778534e-06,
"loss": 0.1984,
"step": 400
},
{
"epoch": 0.31623330990864373,
"grad_norm": 5.261374473571777,
"learning_rate": 9.058956916099774e-06,
"loss": 0.0871,
"step": 450
},
{
"epoch": 0.35137034434293746,
"grad_norm": 0.009863858111202717,
"learning_rate": 8.869992441421013e-06,
"loss": 0.0469,
"step": 500
},
{
"epoch": 0.3865073787772312,
"grad_norm": 0.026721283793449402,
"learning_rate": 8.681027966742253e-06,
"loss": 0.0974,
"step": 550
},
{
"epoch": 0.42164441321152496,
"grad_norm": 0.0059141730889678,
"learning_rate": 8.492063492063492e-06,
"loss": 0.1371,
"step": 600
},
{
"epoch": 0.4567814476458187,
"grad_norm": 0.005736654158681631,
"learning_rate": 8.303099017384732e-06,
"loss": 0.1128,
"step": 650
},
{
"epoch": 0.49191848208011246,
"grad_norm": 0.006507765036076307,
"learning_rate": 8.114134542705972e-06,
"loss": 0.1137,
"step": 700
},
{
"epoch": 0.5270555165144062,
"grad_norm": 3.8664681911468506,
"learning_rate": 7.925170068027211e-06,
"loss": 0.1744,
"step": 750
},
{
"epoch": 0.5621925509486999,
"grad_norm": 0.004181186202913523,
"learning_rate": 7.73620559334845e-06,
"loss": 0.1439,
"step": 800
},
{
"epoch": 0.5973295853829936,
"grad_norm": 1.2505786418914795,
"learning_rate": 7.54724111866969e-06,
"loss": 0.2144,
"step": 850
},
{
"epoch": 0.6324666198172875,
"grad_norm": 0.010104745626449585,
"learning_rate": 7.358276643990931e-06,
"loss": 0.1264,
"step": 900
},
{
"epoch": 0.6676036542515812,
"grad_norm": 0.7443602085113525,
"learning_rate": 7.16931216931217e-06,
"loss": 0.152,
"step": 950
},
{
"epoch": 0.7027406886858749,
"grad_norm": 0.005628188606351614,
"learning_rate": 6.980347694633409e-06,
"loss": 0.1059,
"step": 1000
},
{
"epoch": 0.7378777231201686,
"grad_norm": 1.3959672451019287,
"learning_rate": 6.7913832199546494e-06,
"loss": 0.1033,
"step": 1050
},
{
"epoch": 0.7730147575544624,
"grad_norm": 0.002598799532279372,
"learning_rate": 6.602418745275889e-06,
"loss": 0.102,
"step": 1100
},
{
"epoch": 0.8081517919887562,
"grad_norm": 5.506197452545166,
"learning_rate": 6.413454270597128e-06,
"loss": 0.1057,
"step": 1150
},
{
"epoch": 0.8432888264230499,
"grad_norm": 0.0023803438525646925,
"learning_rate": 6.224489795918368e-06,
"loss": 0.0908,
"step": 1200
},
{
"epoch": 0.8784258608573436,
"grad_norm": 0.14706958830356598,
"learning_rate": 6.035525321239607e-06,
"loss": 0.0753,
"step": 1250
},
{
"epoch": 0.9135628952916374,
"grad_norm": 0.003114320570603013,
"learning_rate": 5.846560846560847e-06,
"loss": 0.1046,
"step": 1300
},
{
"epoch": 0.9486999297259311,
"grad_norm": 0.0024313139729201794,
"learning_rate": 5.657596371882087e-06,
"loss": 0.1181,
"step": 1350
},
{
"epoch": 0.9838369641602249,
"grad_norm": 2.204352378845215,
"learning_rate": 5.468631897203326e-06,
"loss": 0.086,
"step": 1400
}
],
"logging_steps": 50,
"max_steps": 2846,
"num_input_tokens_seen": 0,
"num_train_epochs": 2,
"save_steps": 500,
"stateful_callbacks": {
"TrainerControl": {
"args": {
"should_epoch_stop": false,
"should_evaluate": false,
"should_log": false,
"should_save": true,
"should_training_stop": false
},
"attributes": {}
}
},
"total_flos": 371766674022912.0,
"train_batch_size": 4,
"trial_name": null,
"trial_params": null
}