plip's picture
Training in progress, step 10000
fc976fe
raw
history blame
4.78 kB
{
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.30638193572106986,
"global_step": 10000,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.02,
"learning_rate": 5.999999999999999e-06,
"loss": 0.9345,
"step": 500
},
{
"epoch": 0.03,
"learning_rate": 1.1999999999999999e-05,
"loss": 0.7514,
"step": 1000
},
{
"epoch": 0.03,
"eval_loss": 0.9052248597145081,
"eval_runtime": 0.5073,
"eval_samples_per_second": 1971.284,
"eval_steps_per_second": 31.541,
"step": 1000
},
{
"epoch": 0.05,
"learning_rate": 1.7999999999999997e-05,
"loss": 0.7408,
"step": 1500
},
{
"epoch": 0.06,
"learning_rate": 2.3999999999999997e-05,
"loss": 0.74,
"step": 2000
},
{
"epoch": 0.06,
"eval_loss": 0.9058064818382263,
"eval_runtime": 0.5123,
"eval_samples_per_second": 1952.096,
"eval_steps_per_second": 31.234,
"step": 2000
},
{
"epoch": 0.08,
"learning_rate": 2.9999999999999997e-05,
"loss": 0.7398,
"step": 2500
},
{
"epoch": 0.09,
"learning_rate": 3.5999999999999994e-05,
"loss": 0.7395,
"step": 3000
},
{
"epoch": 0.09,
"eval_loss": 0.9068936705589294,
"eval_runtime": 0.499,
"eval_samples_per_second": 2003.855,
"eval_steps_per_second": 32.062,
"step": 3000
},
{
"epoch": 0.11,
"learning_rate": 4.2e-05,
"loss": 0.7394,
"step": 3500
},
{
"epoch": 0.12,
"learning_rate": 4.7999999999999994e-05,
"loss": 0.7392,
"step": 4000
},
{
"epoch": 0.12,
"eval_loss": 0.9032273888587952,
"eval_runtime": 0.5086,
"eval_samples_per_second": 1966.372,
"eval_steps_per_second": 31.462,
"step": 4000
},
{
"epoch": 0.14,
"learning_rate": 5.399999999999999e-05,
"loss": 0.7389,
"step": 4500
},
{
"epoch": 0.15,
"learning_rate": 5.9999999999999995e-05,
"loss": 0.7386,
"step": 5000
},
{
"epoch": 0.15,
"eval_loss": 0.9034351706504822,
"eval_runtime": 0.5101,
"eval_samples_per_second": 1960.24,
"eval_steps_per_second": 31.364,
"step": 5000
},
{
"epoch": 0.17,
"learning_rate": 6.599999999999999e-05,
"loss": 0.7382,
"step": 5500
},
{
"epoch": 0.18,
"learning_rate": 7.199999999999999e-05,
"loss": 0.7377,
"step": 6000
},
{
"epoch": 0.18,
"eval_loss": 0.8661078810691833,
"eval_runtime": 0.5119,
"eval_samples_per_second": 1953.649,
"eval_steps_per_second": 31.258,
"step": 6000
},
{
"epoch": 0.2,
"learning_rate": 7.8e-05,
"loss": 0.7375,
"step": 6500
},
{
"epoch": 0.21,
"learning_rate": 8.4e-05,
"loss": 0.7373,
"step": 7000
},
{
"epoch": 0.21,
"eval_loss": 0.8658460974693298,
"eval_runtime": 0.5198,
"eval_samples_per_second": 1923.698,
"eval_steps_per_second": 30.779,
"step": 7000
},
{
"epoch": 0.23,
"learning_rate": 8.999999999999999e-05,
"loss": 0.7367,
"step": 7500
},
{
"epoch": 0.25,
"learning_rate": 9.599999999999999e-05,
"loss": 0.7207,
"step": 8000
},
{
"epoch": 0.25,
"eval_loss": 0.8675529956817627,
"eval_runtime": 0.5261,
"eval_samples_per_second": 1900.917,
"eval_steps_per_second": 30.415,
"step": 8000
},
{
"epoch": 0.26,
"learning_rate": 0.000102,
"loss": 0.6905,
"step": 8500
},
{
"epoch": 0.28,
"learning_rate": 0.00010799999999999998,
"loss": 0.6746,
"step": 9000
},
{
"epoch": 0.28,
"eval_loss": 0.8773286938667297,
"eval_runtime": 0.5159,
"eval_samples_per_second": 1938.548,
"eval_steps_per_second": 31.017,
"step": 9000
},
{
"epoch": 0.29,
"learning_rate": 0.00011399999999999999,
"loss": 0.6617,
"step": 9500
},
{
"epoch": 0.31,
"learning_rate": 0.00011999999999999999,
"loss": 0.6406,
"step": 10000
},
{
"epoch": 0.31,
"eval_loss": 0.8803548812866211,
"eval_runtime": 0.5084,
"eval_samples_per_second": 1966.828,
"eval_steps_per_second": 31.469,
"step": 10000
}
],
"max_steps": 500000,
"num_train_epochs": 16,
"total_flos": 3.194871387745e+20,
"trial_name": null,
"trial_params": null
}