| { |
| "best_global_step": null, |
| "best_metric": null, |
| "best_model_checkpoint": null, |
| "epoch": 1.0, |
| "eval_steps": 500, |
| "global_step": 1423, |
| "is_hyper_param_search": false, |
| "is_local_process_zero": true, |
| "is_world_process_zero": true, |
| "log_history": [ |
| { |
| "epoch": 0.035137034434293744, |
| "grad_norm": 13.614160537719727, |
| "learning_rate": 2.4500000000000003e-06, |
| "loss": 0.6868, |
| "step": 50 |
| }, |
| { |
| "epoch": 0.07027406886858749, |
| "grad_norm": 7.1465582847595215, |
| "learning_rate": 4.95e-06, |
| "loss": 0.5593, |
| "step": 100 |
| }, |
| { |
| "epoch": 0.10541110330288124, |
| "grad_norm": 0.4900398254394531, |
| "learning_rate": 7.450000000000001e-06, |
| "loss": 0.545, |
| "step": 150 |
| }, |
| { |
| "epoch": 0.14054813773717498, |
| "grad_norm": 0.034149572253227234, |
| "learning_rate": 9.950000000000001e-06, |
| "loss": 0.2203, |
| "step": 200 |
| }, |
| { |
| "epoch": 0.17568517217146873, |
| "grad_norm": 2.6971633434295654, |
| "learning_rate": 9.814814814814815e-06, |
| "loss": 0.1222, |
| "step": 250 |
| }, |
| { |
| "epoch": 0.21082220660576248, |
| "grad_norm": 0.012744505889713764, |
| "learning_rate": 9.625850340136055e-06, |
| "loss": 0.0908, |
| "step": 300 |
| }, |
| { |
| "epoch": 0.24595924104005623, |
| "grad_norm": 0.008385892026126385, |
| "learning_rate": 9.436885865457295e-06, |
| "loss": 0.0744, |
| "step": 350 |
| }, |
| { |
| "epoch": 0.28109627547434995, |
| "grad_norm": 3.446326494216919, |
| "learning_rate": 9.247921390778534e-06, |
| "loss": 0.1984, |
| "step": 400 |
| }, |
| { |
| "epoch": 0.31623330990864373, |
| "grad_norm": 5.261374473571777, |
| "learning_rate": 9.058956916099774e-06, |
| "loss": 0.0871, |
| "step": 450 |
| }, |
| { |
| "epoch": 0.35137034434293746, |
| "grad_norm": 0.009863858111202717, |
| "learning_rate": 8.869992441421013e-06, |
| "loss": 0.0469, |
| "step": 500 |
| }, |
| { |
| "epoch": 0.3865073787772312, |
| "grad_norm": 0.026721283793449402, |
| "learning_rate": 8.681027966742253e-06, |
| "loss": 0.0974, |
| "step": 550 |
| }, |
| { |
| "epoch": 0.42164441321152496, |
| "grad_norm": 0.0059141730889678, |
| "learning_rate": 8.492063492063492e-06, |
| "loss": 0.1371, |
| "step": 600 |
| }, |
| { |
| "epoch": 0.4567814476458187, |
| "grad_norm": 0.005736654158681631, |
| "learning_rate": 8.303099017384732e-06, |
| "loss": 0.1128, |
| "step": 650 |
| }, |
| { |
| "epoch": 0.49191848208011246, |
| "grad_norm": 0.006507765036076307, |
| "learning_rate": 8.114134542705972e-06, |
| "loss": 0.1137, |
| "step": 700 |
| }, |
| { |
| "epoch": 0.5270555165144062, |
| "grad_norm": 3.8664681911468506, |
| "learning_rate": 7.925170068027211e-06, |
| "loss": 0.1744, |
| "step": 750 |
| }, |
| { |
| "epoch": 0.5621925509486999, |
| "grad_norm": 0.004181186202913523, |
| "learning_rate": 7.73620559334845e-06, |
| "loss": 0.1439, |
| "step": 800 |
| }, |
| { |
| "epoch": 0.5973295853829936, |
| "grad_norm": 1.2505786418914795, |
| "learning_rate": 7.54724111866969e-06, |
| "loss": 0.2144, |
| "step": 850 |
| }, |
| { |
| "epoch": 0.6324666198172875, |
| "grad_norm": 0.010104745626449585, |
| "learning_rate": 7.358276643990931e-06, |
| "loss": 0.1264, |
| "step": 900 |
| }, |
| { |
| "epoch": 0.6676036542515812, |
| "grad_norm": 0.7443602085113525, |
| "learning_rate": 7.16931216931217e-06, |
| "loss": 0.152, |
| "step": 950 |
| }, |
| { |
| "epoch": 0.7027406886858749, |
| "grad_norm": 0.005628188606351614, |
| "learning_rate": 6.980347694633409e-06, |
| "loss": 0.1059, |
| "step": 1000 |
| }, |
| { |
| "epoch": 0.7378777231201686, |
| "grad_norm": 1.3959672451019287, |
| "learning_rate": 6.7913832199546494e-06, |
| "loss": 0.1033, |
| "step": 1050 |
| }, |
| { |
| "epoch": 0.7730147575544624, |
| "grad_norm": 0.002598799532279372, |
| "learning_rate": 6.602418745275889e-06, |
| "loss": 0.102, |
| "step": 1100 |
| }, |
| { |
| "epoch": 0.8081517919887562, |
| "grad_norm": 5.506197452545166, |
| "learning_rate": 6.413454270597128e-06, |
| "loss": 0.1057, |
| "step": 1150 |
| }, |
| { |
| "epoch": 0.8432888264230499, |
| "grad_norm": 0.0023803438525646925, |
| "learning_rate": 6.224489795918368e-06, |
| "loss": 0.0908, |
| "step": 1200 |
| }, |
| { |
| "epoch": 0.8784258608573436, |
| "grad_norm": 0.14706958830356598, |
| "learning_rate": 6.035525321239607e-06, |
| "loss": 0.0753, |
| "step": 1250 |
| }, |
| { |
| "epoch": 0.9135628952916374, |
| "grad_norm": 0.003114320570603013, |
| "learning_rate": 5.846560846560847e-06, |
| "loss": 0.1046, |
| "step": 1300 |
| }, |
| { |
| "epoch": 0.9486999297259311, |
| "grad_norm": 0.0024313139729201794, |
| "learning_rate": 5.657596371882087e-06, |
| "loss": 0.1181, |
| "step": 1350 |
| }, |
| { |
| "epoch": 0.9838369641602249, |
| "grad_norm": 2.204352378845215, |
| "learning_rate": 5.468631897203326e-06, |
| "loss": 0.086, |
| "step": 1400 |
| } |
| ], |
| "logging_steps": 50, |
| "max_steps": 2846, |
| "num_input_tokens_seen": 0, |
| "num_train_epochs": 2, |
| "save_steps": 500, |
| "stateful_callbacks": { |
| "TrainerControl": { |
| "args": { |
| "should_epoch_stop": false, |
| "should_evaluate": false, |
| "should_log": false, |
| "should_save": true, |
| "should_training_stop": false |
| }, |
| "attributes": {} |
| } |
| }, |
| "total_flos": 371766674022912.0, |
| "train_batch_size": 4, |
| "trial_name": null, |
| "trial_params": null |
| } |
|
|