{ "best_global_step": null, "best_metric": null, "best_model_checkpoint": null, "epoch": 1.0, "eval_steps": 500, "global_step": 1423, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.035137034434293744, "grad_norm": 13.614160537719727, "learning_rate": 2.4500000000000003e-06, "loss": 0.6868, "step": 50 }, { "epoch": 0.07027406886858749, "grad_norm": 7.1465582847595215, "learning_rate": 4.95e-06, "loss": 0.5593, "step": 100 }, { "epoch": 0.10541110330288124, "grad_norm": 0.4900398254394531, "learning_rate": 7.450000000000001e-06, "loss": 0.545, "step": 150 }, { "epoch": 0.14054813773717498, "grad_norm": 0.034149572253227234, "learning_rate": 9.950000000000001e-06, "loss": 0.2203, "step": 200 }, { "epoch": 0.17568517217146873, "grad_norm": 2.6971633434295654, "learning_rate": 9.814814814814815e-06, "loss": 0.1222, "step": 250 }, { "epoch": 0.21082220660576248, "grad_norm": 0.012744505889713764, "learning_rate": 9.625850340136055e-06, "loss": 0.0908, "step": 300 }, { "epoch": 0.24595924104005623, "grad_norm": 0.008385892026126385, "learning_rate": 9.436885865457295e-06, "loss": 0.0744, "step": 350 }, { "epoch": 0.28109627547434995, "grad_norm": 3.446326494216919, "learning_rate": 9.247921390778534e-06, "loss": 0.1984, "step": 400 }, { "epoch": 0.31623330990864373, "grad_norm": 5.261374473571777, "learning_rate": 9.058956916099774e-06, "loss": 0.0871, "step": 450 }, { "epoch": 0.35137034434293746, "grad_norm": 0.009863858111202717, "learning_rate": 8.869992441421013e-06, "loss": 0.0469, "step": 500 }, { "epoch": 0.3865073787772312, "grad_norm": 0.026721283793449402, "learning_rate": 8.681027966742253e-06, "loss": 0.0974, "step": 550 }, { "epoch": 0.42164441321152496, "grad_norm": 0.0059141730889678, "learning_rate": 8.492063492063492e-06, "loss": 0.1371, "step": 600 }, { "epoch": 0.4567814476458187, "grad_norm": 0.005736654158681631, "learning_rate": 8.303099017384732e-06, "loss": 0.1128, "step": 650 }, { "epoch": 0.49191848208011246, "grad_norm": 0.006507765036076307, "learning_rate": 8.114134542705972e-06, "loss": 0.1137, "step": 700 }, { "epoch": 0.5270555165144062, "grad_norm": 3.8664681911468506, "learning_rate": 7.925170068027211e-06, "loss": 0.1744, "step": 750 }, { "epoch": 0.5621925509486999, "grad_norm": 0.004181186202913523, "learning_rate": 7.73620559334845e-06, "loss": 0.1439, "step": 800 }, { "epoch": 0.5973295853829936, "grad_norm": 1.2505786418914795, "learning_rate": 7.54724111866969e-06, "loss": 0.2144, "step": 850 }, { "epoch": 0.6324666198172875, "grad_norm": 0.010104745626449585, "learning_rate": 7.358276643990931e-06, "loss": 0.1264, "step": 900 }, { "epoch": 0.6676036542515812, "grad_norm": 0.7443602085113525, "learning_rate": 7.16931216931217e-06, "loss": 0.152, "step": 950 }, { "epoch": 0.7027406886858749, "grad_norm": 0.005628188606351614, "learning_rate": 6.980347694633409e-06, "loss": 0.1059, "step": 1000 }, { "epoch": 0.7378777231201686, "grad_norm": 1.3959672451019287, "learning_rate": 6.7913832199546494e-06, "loss": 0.1033, "step": 1050 }, { "epoch": 0.7730147575544624, "grad_norm": 0.002598799532279372, "learning_rate": 6.602418745275889e-06, "loss": 0.102, "step": 1100 }, { "epoch": 0.8081517919887562, "grad_norm": 5.506197452545166, "learning_rate": 6.413454270597128e-06, "loss": 0.1057, "step": 1150 }, { "epoch": 0.8432888264230499, "grad_norm": 0.0023803438525646925, "learning_rate": 6.224489795918368e-06, "loss": 0.0908, "step": 1200 }, { "epoch": 0.8784258608573436, "grad_norm": 0.14706958830356598, "learning_rate": 6.035525321239607e-06, "loss": 0.0753, "step": 1250 }, { "epoch": 0.9135628952916374, "grad_norm": 0.003114320570603013, "learning_rate": 5.846560846560847e-06, "loss": 0.1046, "step": 1300 }, { "epoch": 0.9486999297259311, "grad_norm": 0.0024313139729201794, "learning_rate": 5.657596371882087e-06, "loss": 0.1181, "step": 1350 }, { "epoch": 0.9838369641602249, "grad_norm": 2.204352378845215, "learning_rate": 5.468631897203326e-06, "loss": 0.086, "step": 1400 } ], "logging_steps": 50, "max_steps": 2846, "num_input_tokens_seen": 0, "num_train_epochs": 2, "save_steps": 500, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": false }, "attributes": {} } }, "total_flos": 371766674022912.0, "train_batch_size": 4, "trial_name": null, "trial_params": null }