perc_240915 / trainer_log.jsonl
3v324v23's picture
First model version
9b222fd
raw
history blame contribute delete
No virus
12.1 kB
{"current_steps": 10, "total_steps": 450, "loss": 1.3973, "learning_rate": 2.222222222222222e-06, "epoch": 0.022222222222222223, "percentage": 2.22, "elapsed_time": "0:00:25", "remaining_time": "0:18:35", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 20, "total_steps": 450, "loss": 1.0847, "learning_rate": 4.444444444444444e-06, "epoch": 0.044444444444444446, "percentage": 4.44, "elapsed_time": "0:00:49", "remaining_time": "0:17:45", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 30, "total_steps": 450, "loss": 1.0589, "learning_rate": 6.666666666666667e-06, "epoch": 0.06666666666666667, "percentage": 6.67, "elapsed_time": "0:01:13", "remaining_time": "0:17:13", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 40, "total_steps": 450, "loss": 1.0995, "learning_rate": 8.888888888888888e-06, "epoch": 0.08888888888888889, "percentage": 8.89, "elapsed_time": "0:01:38", "remaining_time": "0:16:44", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 50, "total_steps": 450, "loss": 1.0963, "learning_rate": 9.996239762521152e-06, "epoch": 0.1111111111111111, "percentage": 11.11, "elapsed_time": "0:02:02", "remaining_time": "0:16:17", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 60, "total_steps": 450, "loss": 1.0545, "learning_rate": 9.966191788709716e-06, "epoch": 0.13333333333333333, "percentage": 13.33, "elapsed_time": "0:02:26", "remaining_time": "0:15:51", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 70, "total_steps": 450, "loss": 1.1422, "learning_rate": 9.906276553136924e-06, "epoch": 0.15555555555555556, "percentage": 15.56, "elapsed_time": "0:02:50", "remaining_time": "0:15:26", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 80, "total_steps": 450, "loss": 1.1892, "learning_rate": 9.816854393079402e-06, "epoch": 0.17777777777777778, "percentage": 17.78, "elapsed_time": "0:03:14", "remaining_time": "0:15:01", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 90, "total_steps": 450, "loss": 1.1044, "learning_rate": 9.698463103929542e-06, "epoch": 0.2, "percentage": 20.0, "elapsed_time": "0:03:39", "remaining_time": "0:14:36", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 100, "total_steps": 450, "loss": 1.0914, "learning_rate": 9.551814704830734e-06, "epoch": 0.2222222222222222, "percentage": 22.22, "elapsed_time": "0:04:03", "remaining_time": "0:14:11", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 100, "total_steps": 450, "eval_loss": 1.1223397254943848, "epoch": 0.2222222222222222, "percentage": 22.22, "elapsed_time": "0:04:40", "remaining_time": "0:16:21", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 110, "total_steps": 450, "loss": 1.1131, "learning_rate": 9.377791156510456e-06, "epoch": 0.24444444444444444, "percentage": 24.44, "elapsed_time": "0:05:04", "remaining_time": "0:15:41", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 120, "total_steps": 450, "loss": 1.0948, "learning_rate": 9.177439057064684e-06, "epoch": 0.26666666666666666, "percentage": 26.67, "elapsed_time": "0:05:28", "remaining_time": "0:15:04", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 130, "total_steps": 450, "loss": 1.1654, "learning_rate": 8.951963347593797e-06, "epoch": 0.28888888888888886, "percentage": 28.89, "elapsed_time": "0:05:53", "remaining_time": "0:14:29", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 140, "total_steps": 450, "loss": 1.0162, "learning_rate": 8.702720065545024e-06, "epoch": 0.3111111111111111, "percentage": 31.11, "elapsed_time": "0:06:17", "remaining_time": "0:13:55", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 150, "total_steps": 450, "loss": 1.1379, "learning_rate": 8.43120818934367e-06, "epoch": 0.3333333333333333, "percentage": 33.33, "elapsed_time": "0:06:41", "remaining_time": "0:13:23", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 160, "total_steps": 450, "loss": 1.1389, "learning_rate": 8.139060623360494e-06, "epoch": 0.35555555555555557, "percentage": 35.56, "elapsed_time": "0:07:05", "remaining_time": "0:12:52", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 170, "total_steps": 450, "loss": 1.066, "learning_rate": 7.828034377432694e-06, "epoch": 0.37777777777777777, "percentage": 37.78, "elapsed_time": "0:07:30", "remaining_time": "0:12:21", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 180, "total_steps": 450, "loss": 1.0525, "learning_rate": 7.500000000000001e-06, "epoch": 0.4, "percentage": 40.0, "elapsed_time": "0:07:54", "remaining_time": "0:11:51", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 190, "total_steps": 450, "loss": 1.0559, "learning_rate": 7.156930328406268e-06, "epoch": 0.4222222222222222, "percentage": 42.22, "elapsed_time": "0:08:18", "remaining_time": "0:11:22", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 200, "total_steps": 450, "loss": 1.1722, "learning_rate": 6.800888624023552e-06, "epoch": 0.4444444444444444, "percentage": 44.44, "elapsed_time": "0:08:42", "remaining_time": "0:10:53", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 200, "total_steps": 450, "eval_loss": 1.0550415515899658, "epoch": 0.4444444444444444, "percentage": 44.44, "elapsed_time": "0:09:19", "remaining_time": "0:11:39", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 210, "total_steps": 450, "loss": 1.1375, "learning_rate": 6.434016163555452e-06, "epoch": 0.4666666666666667, "percentage": 46.67, "elapsed_time": "0:09:43", "remaining_time": "0:11:07", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 220, "total_steps": 450, "loss": 1.1109, "learning_rate": 6.058519361147055e-06, "epoch": 0.4888888888888889, "percentage": 48.89, "elapsed_time": "0:10:08", "remaining_time": "0:10:35", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 230, "total_steps": 450, "loss": 1.0671, "learning_rate": 5.6766564987506564e-06, "epoch": 0.5111111111111111, "percentage": 51.11, "elapsed_time": "0:10:32", "remaining_time": "0:10:04", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 240, "total_steps": 450, "loss": 1.0387, "learning_rate": 5.290724144552379e-06, "epoch": 0.5333333333333333, "percentage": 53.33, "elapsed_time": "0:10:56", "remaining_time": "0:09:34", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 250, "total_steps": 450, "loss": 1.0224, "learning_rate": 4.903043341140879e-06, "epoch": 0.5555555555555556, "percentage": 55.56, "elapsed_time": "0:11:20", "remaining_time": "0:09:04", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 260, "total_steps": 450, "loss": 1.1089, "learning_rate": 4.515945646484105e-06, "epoch": 0.5777777777777777, "percentage": 57.78, "elapsed_time": "0:11:45", "remaining_time": "0:08:35", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 270, "total_steps": 450, "loss": 0.9636, "learning_rate": 4.131759111665349e-06, "epoch": 0.6, "percentage": 60.0, "elapsed_time": "0:12:09", "remaining_time": "0:08:06", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 280, "total_steps": 450, "loss": 1.0456, "learning_rate": 3.752794279710094e-06, "epoch": 0.6222222222222222, "percentage": 62.22, "elapsed_time": "0:12:33", "remaining_time": "0:07:37", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 290, "total_steps": 450, "loss": 0.8803, "learning_rate": 3.3813302897083955e-06, "epoch": 0.6444444444444445, "percentage": 64.44, "elapsed_time": "0:12:57", "remaining_time": "0:07:09", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 300, "total_steps": 450, "loss": 0.9559, "learning_rate": 3.019601169804216e-06, "epoch": 0.6666666666666666, "percentage": 66.67, "elapsed_time": "0:13:22", "remaining_time": "0:06:41", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 300, "total_steps": 450, "eval_loss": 0.9777525067329407, "epoch": 0.6666666666666666, "percentage": 66.67, "elapsed_time": "0:13:58", "remaining_time": "0:06:59", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 310, "total_steps": 450, "loss": 0.9671, "learning_rate": 2.6697824014873076e-06, "epoch": 0.6888888888888889, "percentage": 68.89, "elapsed_time": "0:14:23", "remaining_time": "0:06:29", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 320, "total_steps": 450, "loss": 0.9442, "learning_rate": 2.333977835991545e-06, "epoch": 0.7111111111111111, "percentage": 71.11, "elapsed_time": "0:14:47", "remaining_time": "0:06:00", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 330, "total_steps": 450, "loss": 0.9096, "learning_rate": 2.0142070414860704e-06, "epoch": 0.7333333333333333, "percentage": 73.33, "elapsed_time": "0:15:11", "remaining_time": "0:05:31", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 340, "total_steps": 450, "loss": 0.9629, "learning_rate": 1.7123931571546826e-06, "epoch": 0.7555555555555555, "percentage": 75.56, "elapsed_time": "0:15:35", "remaining_time": "0:05:02", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 350, "total_steps": 450, "loss": 1.0907, "learning_rate": 1.4303513272105057e-06, "epoch": 0.7777777777777778, "percentage": 77.78, "elapsed_time": "0:16:00", "remaining_time": "0:04:34", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 360, "total_steps": 450, "loss": 0.9574, "learning_rate": 1.1697777844051105e-06, "epoch": 0.8, "percentage": 80.0, "elapsed_time": "0:16:24", "remaining_time": "0:04:06", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 370, "total_steps": 450, "loss": 0.9702, "learning_rate": 9.322396486851626e-07, "epoch": 0.8222222222222222, "percentage": 82.22, "elapsed_time": "0:16:48", "remaining_time": "0:03:38", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 380, "total_steps": 450, "loss": 0.9496, "learning_rate": 7.191655023486682e-07, "epoch": 0.8444444444444444, "percentage": 84.44, "elapsed_time": "0:17:12", "remaining_time": "0:03:10", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 390, "total_steps": 450, "loss": 0.8602, "learning_rate": 5.318367983829393e-07, "epoch": 0.8666666666666667, "percentage": 86.67, "elapsed_time": "0:17:36", "remaining_time": "0:02:42", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 400, "total_steps": 450, "loss": 0.9108, "learning_rate": 3.7138015365554834e-07, "epoch": 0.8888888888888888, "percentage": 88.89, "elapsed_time": "0:18:01", "remaining_time": "0:02:15", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 400, "total_steps": 450, "eval_loss": 0.9367556571960449, "epoch": 0.8888888888888888, "percentage": 88.89, "elapsed_time": "0:18:37", "remaining_time": "0:02:19", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 410, "total_steps": 450, "loss": 0.9506, "learning_rate": 2.3876057330792344e-07, "epoch": 0.9111111111111111, "percentage": 91.11, "elapsed_time": "0:19:02", "remaining_time": "0:01:51", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 420, "total_steps": 450, "loss": 0.8663, "learning_rate": 1.3477564710088097e-07, "epoch": 0.9333333333333333, "percentage": 93.33, "elapsed_time": "0:19:26", "remaining_time": "0:01:23", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 430, "total_steps": 450, "loss": 0.9494, "learning_rate": 6.005075261595495e-08, "epoch": 0.9555555555555556, "percentage": 95.56, "elapsed_time": "0:19:50", "remaining_time": "0:00:55", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 440, "total_steps": 450, "loss": 0.9524, "learning_rate": 1.5035294161039882e-08, "epoch": 0.9777777777777777, "percentage": 97.78, "elapsed_time": "0:20:14", "remaining_time": "0:00:27", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 450, "total_steps": 450, "loss": 0.9986, "learning_rate": 0.0, "epoch": 1.0, "percentage": 100.0, "elapsed_time": "0:20:39", "remaining_time": "0:00:00", "throughput": "0.00", "total_tokens": 0}
{"current_steps": 450, "total_steps": 450, "epoch": 1.0, "percentage": 100.0, "elapsed_time": "0:20:39", "remaining_time": "0:00:00", "throughput": "0.00", "total_tokens": 0}