diff --git a/.gitattributes b/.gitattributes
index a6344aac8c09253b3b630fb776ae94478aa0275b..a8561e1e175ac1deb33f1cef01dc2ee9652231ba 100644
--- a/.gitattributes
+++ b/.gitattributes
@@ -33,3 +33,18 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+. filter=lfs diff=lfs merge=lfs -text
+config.json filter=lfs diff=lfs merge=lfs -text
+model-00004-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
+special_tokens_map.json filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-308 filter=lfs diff=lfs merge=lfs -text
+generation_config.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-100 filter=lfs diff=lfs merge=lfs -text
+model-00001-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
+tokenizer_config.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-200 filter=lfs diff=lfs merge=lfs -text
+checkpoint-300 filter=lfs diff=lfs merge=lfs -text
+model-00002-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
+model-00003-of-00004.safetensors filter=lfs diff=lfs merge=lfs -text
+model.safetensors.index.json filter=lfs diff=lfs merge=lfs -text
diff --git a/checkpoint-100/config.json b/checkpoint-100/config.json
new file mode 100644
index 0000000000000000000000000000000000000000..de55730a672c2805e1108109d81a111b0cdb48b1
--- /dev/null
+++ b/checkpoint-100/config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f8b621112eddc9da772fae3f70d672f4d949cc5cb1d580f135d66b3836598c1a
+size 915
diff --git a/checkpoint-100/generation_config.json b/checkpoint-100/generation_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..1007819213eaa3e36dd0c6354016734419f4a863
--- /dev/null
+++ b/checkpoint-100/generation_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a7327788370b7783e29852cf19b4a1f40cd31af5962c94bba76c336d491b4169
+size 184
diff --git a/checkpoint-100/model-00001-of-00004.safetensors b/checkpoint-100/model-00001-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..a1312e33b2bdaab375f7e25ac9e1b0f5817eafc2
--- /dev/null
+++ b/checkpoint-100/model-00001-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a62275b83768b450eaceec040629b66fcc8362137aba5e874d5334c846ac7646
+size 4976698672
diff --git a/checkpoint-100/model-00002-of-00004.safetensors b/checkpoint-100/model-00002-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..9dba2a4b6e065c94c5a37aa9d6d1d44d88cebc1b
--- /dev/null
+++ b/checkpoint-100/model-00002-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cb5b0664bdb88ef83ded2a10b6b37a3517a9db7e0039a688e7351577a3a3a2ed
+size 4999802720
diff --git a/checkpoint-100/model-00003-of-00004.safetensors b/checkpoint-100/model-00003-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3574cdb53e9103641f48bb9ee516b78719f07f1b
--- /dev/null
+++ b/checkpoint-100/model-00003-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8726c320884ad49027dbafeaa3db55a378cf5b1b286b3a02bfa6075988cd760b
+size 4915916176
diff --git a/checkpoint-100/model-00004-of-00004.safetensors b/checkpoint-100/model-00004-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..df8a62ad868b407fd1edf4b19977870e020f4282
--- /dev/null
+++ b/checkpoint-100/model-00004-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4dc1f3d37eb965816560312e02d7a7aaac58b22e1978650cb76b59aaa1ab86df
+size 1168138808
diff --git a/checkpoint-100/model.safetensors.index.json b/checkpoint-100/model.safetensors.index.json
new file mode 100644
index 0000000000000000000000000000000000000000..a054aad8cf0cb5671e2e10b3bf817725a4bec031
--- /dev/null
+++ b/checkpoint-100/model.safetensors.index.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:146776fce3f6db1103aa6f249e65ee5544c5923ce6f971b092eee79aa6e5d37b
+size 23950
diff --git a/checkpoint-100/optimizer.pt b/checkpoint-100/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..e65bfd5175e229465f1a0b950c76922c641008d0
--- /dev/null
+++ b/checkpoint-100/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8e14320965875e72c6d9efb454e3129b26c88bee6e66cda297c6d78702a4f79a
+size 32121299754
diff --git a/checkpoint-100/rng_state.pth b/checkpoint-100/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..4f427f074dcc36845a8b146ca04d57ca91cc33c9
--- /dev/null
+++ b/checkpoint-100/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c26b481b9cb8c27b6ce66e286d7b2720e0aefc60735ac6e7640899dd50d4bf55
+size 14244
diff --git a/checkpoint-100/scheduler.pt b/checkpoint-100/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..2c5eaf5575ce5b30f41aa28327c5f0c94a456719
--- /dev/null
+++ b/checkpoint-100/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:190d296d912b9aa1e1918db9af8ef09674f2d36d1e112918274e76c09503d1c5
+size 1064
diff --git a/checkpoint-100/special_tokens_map.json b/checkpoint-100/special_tokens_map.json
new file mode 100644
index 0000000000000000000000000000000000000000..92d7fab6729d723f45689803b99006fec945cc47
--- /dev/null
+++ b/checkpoint-100/special_tokens_map.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b1835caa5b4d70acaa210fa222b0036f1882f9525c4660fd4810fb3e1e40ff8
+size 325
diff --git a/checkpoint-100/tokenizer.json b/checkpoint-100/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..66cd9d7e0daec95eb10d16a63c615637dbbb7304
--- /dev/null
+++ b/checkpoint-100/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79e3e522635f3171300913bb421464a87de6222182a0570b9b2ccba2a964b2b4
+size 9085657
diff --git a/checkpoint-100/tokenizer_config.json b/checkpoint-100/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..08ebc9598183a9a76c44ef06813aca75f51de4a5
--- /dev/null
+++ b/checkpoint-100/tokenizer_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:05d0de6d831b5940c3f2eae4e495b9f77612cbb05b5804846733ab617f0dfe9b
+size 55380
diff --git a/checkpoint-100/trainer_state.json b/checkpoint-100/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..74d109d1ad4683f913041e596f18ef28d46a9b0b
--- /dev/null
+++ b/checkpoint-100/trainer_state.json
@@ -0,0 +1,823 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.3246753246753247,
+  "eval_steps": 10,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.003246753246753247,
+      "grad_norm": 16.375,
+      "learning_rate": 4.0000000000000003e-07,
+      "loss": 1.4017,
+      "step": 1
+    },
+    {
+      "epoch": 0.006493506493506494,
+      "grad_norm": 16.25,
+      "learning_rate": 8.000000000000001e-07,
+      "loss": 1.4229,
+      "step": 2
+    },
+    {
+      "epoch": 0.00974025974025974,
+      "grad_norm": 16.75,
+      "learning_rate": 1.2000000000000002e-06,
+      "loss": 1.4112,
+      "step": 3
+    },
+    {
+      "epoch": 0.012987012987012988,
+      "grad_norm": 16.5,
+      "learning_rate": 1.6000000000000001e-06,
+      "loss": 1.3959,
+      "step": 4
+    },
+    {
+      "epoch": 0.016233766233766232,
+      "grad_norm": 16.125,
+      "learning_rate": 2.0000000000000003e-06,
+      "loss": 1.3789,
+      "step": 5
+    },
+    {
+      "epoch": 0.01948051948051948,
+      "grad_norm": 14.4375,
+      "learning_rate": 2.4000000000000003e-06,
+      "loss": 1.3618,
+      "step": 6
+    },
+    {
+      "epoch": 0.022727272727272728,
+      "grad_norm": 10.1875,
+      "learning_rate": 2.8000000000000003e-06,
+      "loss": 1.3214,
+      "step": 7
+    },
+    {
+      "epoch": 0.025974025974025976,
+      "grad_norm": 7.4375,
+      "learning_rate": 3.2000000000000003e-06,
+      "loss": 1.2885,
+      "step": 8
+    },
+    {
+      "epoch": 0.02922077922077922,
+      "grad_norm": 6.28125,
+      "learning_rate": 3.6000000000000003e-06,
+      "loss": 1.2436,
+      "step": 9
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "grad_norm": 5.75,
+      "learning_rate": 4.000000000000001e-06,
+      "loss": 1.2269,
+      "step": 10
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "eval_loss": 1.1976113319396973,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0273,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 10
+    },
+    {
+      "epoch": 0.03571428571428571,
+      "grad_norm": 5.375,
+      "learning_rate": 4.4e-06,
+      "loss": 1.1925,
+      "step": 11
+    },
+    {
+      "epoch": 0.03896103896103896,
+      "grad_norm": 5.125,
+      "learning_rate": 4.800000000000001e-06,
+      "loss": 1.1307,
+      "step": 12
+    },
+    {
+      "epoch": 0.04220779220779221,
+      "grad_norm": 5.125,
+      "learning_rate": 5.2e-06,
+      "loss": 1.0778,
+      "step": 13
+    },
+    {
+      "epoch": 0.045454545454545456,
+      "grad_norm": 5.0625,
+      "learning_rate": 5.600000000000001e-06,
+      "loss": 1.0067,
+      "step": 14
+    },
+    {
+      "epoch": 0.048701298701298704,
+      "grad_norm": 5.75,
+      "learning_rate": 6e-06,
+      "loss": 0.9668,
+      "step": 15
+    },
+    {
+      "epoch": 0.05194805194805195,
+      "grad_norm": 5.15625,
+      "learning_rate": 6.4000000000000006e-06,
+      "loss": 0.875,
+      "step": 16
+    },
+    {
+      "epoch": 0.05519480519480519,
+      "grad_norm": 4.875,
+      "learning_rate": 6.800000000000001e-06,
+      "loss": 0.8297,
+      "step": 17
+    },
+    {
+      "epoch": 0.05844155844155844,
+      "grad_norm": 4.5625,
+      "learning_rate": 7.2000000000000005e-06,
+      "loss": 0.7703,
+      "step": 18
+    },
+    {
+      "epoch": 0.06168831168831169,
+      "grad_norm": 3.90625,
+      "learning_rate": 7.600000000000001e-06,
+      "loss": 0.7327,
+      "step": 19
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "grad_norm": 2.78125,
+      "learning_rate": 8.000000000000001e-06,
+      "loss": 0.6672,
+      "step": 20
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "eval_loss": 0.6521233320236206,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.999,
+      "eval_samples_per_second": 11.88,
+      "eval_steps_per_second": 1.486,
+      "step": 20
+    },
+    {
+      "epoch": 0.06818181818181818,
+      "grad_norm": 2.140625,
+      "learning_rate": 8.400000000000001e-06,
+      "loss": 0.6507,
+      "step": 21
+    },
+    {
+      "epoch": 0.07142857142857142,
+      "grad_norm": 1.828125,
+      "learning_rate": 8.8e-06,
+      "loss": 0.6215,
+      "step": 22
+    },
+    {
+      "epoch": 0.07467532467532467,
+      "grad_norm": 1.5546875,
+      "learning_rate": 9.200000000000002e-06,
+      "loss": 0.5992,
+      "step": 23
+    },
+    {
+      "epoch": 0.07792207792207792,
+      "grad_norm": 1.4453125,
+      "learning_rate": 9.600000000000001e-06,
+      "loss": 0.5748,
+      "step": 24
+    },
+    {
+      "epoch": 0.08116883116883117,
+      "grad_norm": 1.328125,
+      "learning_rate": 1e-05,
+      "loss": 0.5572,
+      "step": 25
+    },
+    {
+      "epoch": 0.08441558441558442,
+      "grad_norm": 1.4765625,
+      "learning_rate": 1.04e-05,
+      "loss": 0.542,
+      "step": 26
+    },
+    {
+      "epoch": 0.08766233766233766,
+      "grad_norm": 1.1171875,
+      "learning_rate": 1.0800000000000002e-05,
+      "loss": 0.5317,
+      "step": 27
+    },
+    {
+      "epoch": 0.09090909090909091,
+      "grad_norm": 0.90625,
+      "learning_rate": 1.1200000000000001e-05,
+      "loss": 0.5173,
+      "step": 28
+    },
+    {
+      "epoch": 0.09415584415584416,
+      "grad_norm": 0.83984375,
+      "learning_rate": 1.16e-05,
+      "loss": 0.4892,
+      "step": 29
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "grad_norm": 0.828125,
+      "learning_rate": 1.2e-05,
+      "loss": 0.491,
+      "step": 30
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "eval_loss": 0.48583006858825684,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9705,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 30
+    },
+    {
+      "epoch": 0.10064935064935066,
+      "grad_norm": 0.8359375,
+      "learning_rate": 1.2400000000000002e-05,
+      "loss": 0.4852,
+      "step": 31
+    },
+    {
+      "epoch": 0.1038961038961039,
+      "grad_norm": 1.0078125,
+      "learning_rate": 1.2800000000000001e-05,
+      "loss": 0.4749,
+      "step": 32
+    },
+    {
+      "epoch": 0.10714285714285714,
+      "grad_norm": 0.99609375,
+      "learning_rate": 1.3200000000000002e-05,
+      "loss": 0.467,
+      "step": 33
+    },
+    {
+      "epoch": 0.11038961038961038,
+      "grad_norm": 0.734375,
+      "learning_rate": 1.3600000000000002e-05,
+      "loss": 0.4545,
+      "step": 34
+    },
+    {
+      "epoch": 0.11363636363636363,
+      "grad_norm": 0.69140625,
+      "learning_rate": 1.4e-05,
+      "loss": 0.4609,
+      "step": 35
+    },
+    {
+      "epoch": 0.11688311688311688,
+      "grad_norm": 0.69921875,
+      "learning_rate": 1.4400000000000001e-05,
+      "loss": 0.4497,
+      "step": 36
+    },
+    {
+      "epoch": 0.12012987012987013,
+      "grad_norm": 0.703125,
+      "learning_rate": 1.48e-05,
+      "loss": 0.4607,
+      "step": 37
+    },
+    {
+      "epoch": 0.12337662337662338,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5200000000000002e-05,
+      "loss": 0.443,
+      "step": 38
+    },
+    {
+      "epoch": 0.1266233766233766,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.5600000000000003e-05,
+      "loss": 0.4428,
+      "step": 39
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.6000000000000003e-05,
+      "loss": 0.435,
+      "step": 40
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "eval_loss": 0.4367639422416687,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0177,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 40
+    },
+    {
+      "epoch": 0.1331168831168831,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.64e-05,
+      "loss": 0.438,
+      "step": 41
+    },
+    {
+      "epoch": 0.13636363636363635,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.6800000000000002e-05,
+      "loss": 0.4392,
+      "step": 42
+    },
+    {
+      "epoch": 0.1396103896103896,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.72e-05,
+      "loss": 0.435,
+      "step": 43
+    },
+    {
+      "epoch": 0.14285714285714285,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.76e-05,
+      "loss": 0.4356,
+      "step": 44
+    },
+    {
+      "epoch": 0.1461038961038961,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.8e-05,
+      "loss": 0.4172,
+      "step": 45
+    },
+    {
+      "epoch": 0.14935064935064934,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.8400000000000003e-05,
+      "loss": 0.4162,
+      "step": 46
+    },
+    {
+      "epoch": 0.1525974025974026,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.88e-05,
+      "loss": 0.418,
+      "step": 47
+    },
+    {
+      "epoch": 0.15584415584415584,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.9200000000000003e-05,
+      "loss": 0.4177,
+      "step": 48
+    },
+    {
+      "epoch": 0.1590909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9600000000000002e-05,
+      "loss": 0.4064,
+      "step": 49
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "grad_norm": 0.62890625,
+      "learning_rate": 2e-05,
+      "loss": 0.4078,
+      "step": 50
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "eval_loss": 0.4077843129634857,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.8482,
+      "eval_samples_per_second": 11.89,
+      "eval_steps_per_second": 1.487,
+      "step": 50
+    },
+    {
+      "epoch": 0.16558441558441558,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9999258647132645e-05,
+      "loss": 0.4057,
+      "step": 51
+    },
+    {
+      "epoch": 0.16883116883116883,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.9997034698451396e-05,
+      "loss": 0.4172,
+      "step": 52
+    },
+    {
+      "epoch": 0.17207792207792208,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9993328483702393e-05,
+      "loss": 0.4082,
+      "step": 53
+    },
+    {
+      "epoch": 0.17532467532467533,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998814055240823e-05,
+      "loss": 0.3973,
+      "step": 54
+    },
+    {
+      "epoch": 0.17857142857142858,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998147167378645e-05,
+      "loss": 0.3969,
+      "step": 55
+    },
+    {
+      "epoch": 0.18181818181818182,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9973322836635517e-05,
+      "loss": 0.4032,
+      "step": 56
+    },
+    {
+      "epoch": 0.18506493506493507,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9963695249188185e-05,
+      "loss": 0.3956,
+      "step": 57
+    },
+    {
+      "epoch": 0.18831168831168832,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.995259033893236e-05,
+      "loss": 0.3952,
+      "step": 58
+    },
+    {
+      "epoch": 0.19155844155844157,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9940009752399462e-05,
+      "loss": 0.3905,
+      "step": 59
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9925955354920265e-05,
+      "loss": 0.3822,
+      "step": 60
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "eval_loss": 0.38575297594070435,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9122,
+      "eval_samples_per_second": 11.886,
+      "eval_steps_per_second": 1.486,
+      "step": 60
+    },
+    {
+      "epoch": 0.19805194805194806,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9910429230348348e-05,
+      "loss": 0.3976,
+      "step": 61
+    },
+    {
+      "epoch": 0.2012987012987013,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9893433680751105e-05,
+      "loss": 0.3714,
+      "step": 62
+    },
+    {
+      "epoch": 0.20454545454545456,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9874971226068417e-05,
+      "loss": 0.3739,
+      "step": 63
+    },
+    {
+      "epoch": 0.2077922077922078,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.985504460373903e-05,
+      "loss": 0.3907,
+      "step": 64
+    },
+    {
+      "epoch": 0.21103896103896103,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.983365676829466e-05,
+      "loss": 0.3803,
+      "step": 65
+    },
+    {
+      "epoch": 0.21428571428571427,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9810810890921943e-05,
+      "loss": 0.3902,
+      "step": 66
+    },
+    {
+      "epoch": 0.21753246753246752,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9786510358992213e-05,
+      "loss": 0.3744,
+      "step": 67
+    },
+    {
+      "epoch": 0.22077922077922077,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.9760758775559275e-05,
+      "loss": 0.3668,
+      "step": 68
+    },
+    {
+      "epoch": 0.22402597402597402,
+      "grad_norm": 0.625,
+      "learning_rate": 1.9733559958825167e-05,
+      "loss": 0.3755,
+      "step": 69
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9704917941574053e-05,
+      "loss": 0.3733,
+      "step": 70
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "eval_loss": 0.3694484233856201,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9363,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 70
+    },
+    {
+      "epoch": 0.2305194805194805,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.9674836970574253e-05,
+      "loss": 0.3676,
+      "step": 71
+    },
+    {
+      "epoch": 0.23376623376623376,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9643321505948588e-05,
+      "loss": 0.3721,
+      "step": 72
+    },
+    {
+      "epoch": 0.237012987012987,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9610376220513067e-05,
+      "loss": 0.3558,
+      "step": 73
+    },
+    {
+      "epoch": 0.24025974025974026,
+      "grad_norm": 0.625,
+      "learning_rate": 1.957600599908406e-05,
+      "loss": 0.357,
+      "step": 74
+    },
+    {
+      "epoch": 0.2435064935064935,
+      "grad_norm": 0.66015625,
+      "learning_rate": 1.954021593775401e-05,
+      "loss": 0.3608,
+      "step": 75
+    },
+    {
+      "epoch": 0.24675324675324675,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9503011343135828e-05,
+      "loss": 0.3513,
+      "step": 76
+    },
+    {
+      "epoch": 0.25,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9464397731576093e-05,
+      "loss": 0.3693,
+      "step": 77
+    },
+    {
+      "epoch": 0.2532467532467532,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9424380828337146e-05,
+      "loss": 0.3605,
+      "step": 78
+    },
+    {
+      "epoch": 0.2564935064935065,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.938296656674817e-05,
+      "loss": 0.3553,
+      "step": 79
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.9340161087325483e-05,
+      "loss": 0.3721,
+      "step": 80
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "eval_loss": 0.3553767800331116,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0277,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 80
+    },
+    {
+      "epoch": 0.262987012987013,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9295970736862063e-05,
+      "loss": 0.3597,
+      "step": 81
+    },
+    {
+      "epoch": 0.2662337662337662,
+      "grad_norm": 0.65625,
+      "learning_rate": 1.9250402067486523e-05,
+      "loss": 0.3527,
+      "step": 82
+    },
+    {
+      "epoch": 0.2694805194805195,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9203461835691596e-05,
+      "loss": 0.3528,
+      "step": 83
+    },
+    {
+      "epoch": 0.2727272727272727,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.9155157001332374e-05,
+      "loss": 0.356,
+      "step": 84
+    },
+    {
+      "epoch": 0.275974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9105494726594344e-05,
+      "loss": 0.3556,
+      "step": 85
+    },
+    {
+      "epoch": 0.2792207792207792,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.905448237493147e-05,
+      "loss": 0.3473,
+      "step": 86
+    },
+    {
+      "epoch": 0.2824675324675325,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9002127509974376e-05,
+      "loss": 0.3441,
+      "step": 87
+    },
+    {
+      "epoch": 0.2857142857142857,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.894843789440892e-05,
+      "loss": 0.3492,
+      "step": 88
+    },
+    {
+      "epoch": 0.288961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.889342148882519e-05,
+      "loss": 0.3338,
+      "step": 89
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "grad_norm": 0.67578125,
+      "learning_rate": 1.8837086450537195e-05,
+      "loss": 0.3347,
+      "step": 90
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "eval_loss": 0.34332525730133057,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9257,
+      "eval_samples_per_second": 11.885,
+      "eval_steps_per_second": 1.486,
+      "step": 90
+    },
+    {
+      "epoch": 0.29545454545454547,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.877944113237336e-05,
+      "loss": 0.3438,
+      "step": 91
+    },
+    {
+      "epoch": 0.2987012987012987,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.872049408143808e-05,
+      "loss": 0.3505,
+      "step": 92
+    },
+    {
+      "epoch": 0.30194805194805197,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.866025403784439e-05,
+      "loss": 0.3362,
+      "step": 93
+    },
+    {
+      "epoch": 0.3051948051948052,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.8598729933418102e-05,
+      "loss": 0.337,
+      "step": 94
+    },
+    {
+      "epoch": 0.30844155844155846,
+      "grad_norm": 0.64453125,
+      "learning_rate": 1.8535930890373467e-05,
+      "loss": 0.3324,
+      "step": 95
+    },
+    {
+      "epoch": 0.3116883116883117,
+      "grad_norm": 0.6875,
+      "learning_rate": 1.8471866219960604e-05,
+      "loss": 0.3461,
+      "step": 96
+    },
+    {
+      "epoch": 0.31493506493506496,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.840654542108494e-05,
+      "loss": 0.3318,
+      "step": 97
+    },
+    {
+      "epoch": 0.3181818181818182,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.833997817889878e-05,
+      "loss": 0.3335,
+      "step": 98
+    },
+    {
+      "epoch": 0.32142857142857145,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.82721743633653e-05,
+      "loss": 0.3291,
+      "step": 99
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.820314402779511e-05,
+      "loss": 0.3324,
+      "step": 100
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "eval_loss": 0.33332687616348267,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9506,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 100
+    }
+  ],
+  "logging_steps": 1,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 100,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 5.902112919650304e+17,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/checkpoint-100/training_args.bin b/checkpoint-100/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..4684d3c90558d6bfde405c82b58596bc6231bfee
--- /dev/null
+++ b/checkpoint-100/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d09f69be12f7ef607f0a1508435a517f5277e3160e5d6a59346e619fcbaa3592
+size 5176
diff --git a/checkpoint-200/config.json b/checkpoint-200/config.json
new file mode 100644
index 0000000000000000000000000000000000000000..de55730a672c2805e1108109d81a111b0cdb48b1
--- /dev/null
+++ b/checkpoint-200/config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f8b621112eddc9da772fae3f70d672f4d949cc5cb1d580f135d66b3836598c1a
+size 915
diff --git a/checkpoint-200/generation_config.json b/checkpoint-200/generation_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..1007819213eaa3e36dd0c6354016734419f4a863
--- /dev/null
+++ b/checkpoint-200/generation_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a7327788370b7783e29852cf19b4a1f40cd31af5962c94bba76c336d491b4169
+size 184
diff --git a/checkpoint-200/model-00001-of-00004.safetensors b/checkpoint-200/model-00001-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..b93fb6a70ec884fefa8fb9e512f28014f307c5d3
--- /dev/null
+++ b/checkpoint-200/model-00001-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f60fb798be6dc6cc800be220165509b7026f9d04bd5a45d6fae8c0cf42cbf3dd
+size 4976698672
diff --git a/checkpoint-200/model-00002-of-00004.safetensors b/checkpoint-200/model-00002-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..13989d38402ddcfbe36426f1055a88abed496fcc
--- /dev/null
+++ b/checkpoint-200/model-00002-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e1161b03770b7cceaea53010a7c38cae4279bc7ba02d9c0f3584d7574226c2f1
+size 4999802720
diff --git a/checkpoint-200/model-00003-of-00004.safetensors b/checkpoint-200/model-00003-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..e18db12e3a7135207816b0dd636a61b5e3f99d02
--- /dev/null
+++ b/checkpoint-200/model-00003-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:591b6342c93199d69b2d7dbe15e08a0975ac4bdcc33b07e0b5b542e8b6443413
+size 4915916176
diff --git a/checkpoint-200/model-00004-of-00004.safetensors b/checkpoint-200/model-00004-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..217828132fe10ef3484b01ebe969c47246643b1f
--- /dev/null
+++ b/checkpoint-200/model-00004-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2ee9a2f97c92c6b4160444f1a13d52f249d6116c60b9952ae9ad61c992f57f53
+size 1168138808
diff --git a/checkpoint-200/model.safetensors.index.json b/checkpoint-200/model.safetensors.index.json
new file mode 100644
index 0000000000000000000000000000000000000000..a054aad8cf0cb5671e2e10b3bf817725a4bec031
--- /dev/null
+++ b/checkpoint-200/model.safetensors.index.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:146776fce3f6db1103aa6f249e65ee5544c5923ce6f971b092eee79aa6e5d37b
+size 23950
diff --git a/checkpoint-200/optimizer.pt b/checkpoint-200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..adb85a97d34e21fd969fd25f6211f6315fd4497a
--- /dev/null
+++ b/checkpoint-200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:025f20483fbcef43f00f5dd8acbb4ab663455020f9b7ec9b2086cfadf235d021
+size 32121299754
diff --git a/checkpoint-200/rng_state.pth b/checkpoint-200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..ea17f6c5f674e0024fa4c4dc7b8e9d3161488eb2
--- /dev/null
+++ b/checkpoint-200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aad5da87704f9496a4cad943d2aa8988a987b1e77b57fa4a28ff4c89ee480ac0
+size 14244
diff --git a/checkpoint-200/scheduler.pt b/checkpoint-200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..dcaef4e2ef2966c7ebc24289d3f1cf775db9fecb
--- /dev/null
+++ b/checkpoint-200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4565f7ec171134b011efb3a86c834a370c710414c88a26b6961aac2a0ed0d92d
+size 1064
diff --git a/checkpoint-200/special_tokens_map.json b/checkpoint-200/special_tokens_map.json
new file mode 100644
index 0000000000000000000000000000000000000000..92d7fab6729d723f45689803b99006fec945cc47
--- /dev/null
+++ b/checkpoint-200/special_tokens_map.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b1835caa5b4d70acaa210fa222b0036f1882f9525c4660fd4810fb3e1e40ff8
+size 325
diff --git a/checkpoint-200/tokenizer.json b/checkpoint-200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..66cd9d7e0daec95eb10d16a63c615637dbbb7304
--- /dev/null
+++ b/checkpoint-200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79e3e522635f3171300913bb421464a87de6222182a0570b9b2ccba2a964b2b4
+size 9085657
diff --git a/checkpoint-200/tokenizer_config.json b/checkpoint-200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..08ebc9598183a9a76c44ef06813aca75f51de4a5
--- /dev/null
+++ b/checkpoint-200/tokenizer_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:05d0de6d831b5940c3f2eae4e495b9f77612cbb05b5804846733ab617f0dfe9b
+size 55380
diff --git a/checkpoint-200/trainer_state.json b/checkpoint-200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..12d700d76ba184fe64e59008829d6815d563c6db
--- /dev/null
+++ b/checkpoint-200/trainer_state.json
@@ -0,0 +1,1613 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.6493506493506493,
+  "eval_steps": 10,
+  "global_step": 200,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.003246753246753247,
+      "grad_norm": 16.375,
+      "learning_rate": 4.0000000000000003e-07,
+      "loss": 1.4017,
+      "step": 1
+    },
+    {
+      "epoch": 0.006493506493506494,
+      "grad_norm": 16.25,
+      "learning_rate": 8.000000000000001e-07,
+      "loss": 1.4229,
+      "step": 2
+    },
+    {
+      "epoch": 0.00974025974025974,
+      "grad_norm": 16.75,
+      "learning_rate": 1.2000000000000002e-06,
+      "loss": 1.4112,
+      "step": 3
+    },
+    {
+      "epoch": 0.012987012987012988,
+      "grad_norm": 16.5,
+      "learning_rate": 1.6000000000000001e-06,
+      "loss": 1.3959,
+      "step": 4
+    },
+    {
+      "epoch": 0.016233766233766232,
+      "grad_norm": 16.125,
+      "learning_rate": 2.0000000000000003e-06,
+      "loss": 1.3789,
+      "step": 5
+    },
+    {
+      "epoch": 0.01948051948051948,
+      "grad_norm": 14.4375,
+      "learning_rate": 2.4000000000000003e-06,
+      "loss": 1.3618,
+      "step": 6
+    },
+    {
+      "epoch": 0.022727272727272728,
+      "grad_norm": 10.1875,
+      "learning_rate": 2.8000000000000003e-06,
+      "loss": 1.3214,
+      "step": 7
+    },
+    {
+      "epoch": 0.025974025974025976,
+      "grad_norm": 7.4375,
+      "learning_rate": 3.2000000000000003e-06,
+      "loss": 1.2885,
+      "step": 8
+    },
+    {
+      "epoch": 0.02922077922077922,
+      "grad_norm": 6.28125,
+      "learning_rate": 3.6000000000000003e-06,
+      "loss": 1.2436,
+      "step": 9
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "grad_norm": 5.75,
+      "learning_rate": 4.000000000000001e-06,
+      "loss": 1.2269,
+      "step": 10
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "eval_loss": 1.1976113319396973,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0273,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 10
+    },
+    {
+      "epoch": 0.03571428571428571,
+      "grad_norm": 5.375,
+      "learning_rate": 4.4e-06,
+      "loss": 1.1925,
+      "step": 11
+    },
+    {
+      "epoch": 0.03896103896103896,
+      "grad_norm": 5.125,
+      "learning_rate": 4.800000000000001e-06,
+      "loss": 1.1307,
+      "step": 12
+    },
+    {
+      "epoch": 0.04220779220779221,
+      "grad_norm": 5.125,
+      "learning_rate": 5.2e-06,
+      "loss": 1.0778,
+      "step": 13
+    },
+    {
+      "epoch": 0.045454545454545456,
+      "grad_norm": 5.0625,
+      "learning_rate": 5.600000000000001e-06,
+      "loss": 1.0067,
+      "step": 14
+    },
+    {
+      "epoch": 0.048701298701298704,
+      "grad_norm": 5.75,
+      "learning_rate": 6e-06,
+      "loss": 0.9668,
+      "step": 15
+    },
+    {
+      "epoch": 0.05194805194805195,
+      "grad_norm": 5.15625,
+      "learning_rate": 6.4000000000000006e-06,
+      "loss": 0.875,
+      "step": 16
+    },
+    {
+      "epoch": 0.05519480519480519,
+      "grad_norm": 4.875,
+      "learning_rate": 6.800000000000001e-06,
+      "loss": 0.8297,
+      "step": 17
+    },
+    {
+      "epoch": 0.05844155844155844,
+      "grad_norm": 4.5625,
+      "learning_rate": 7.2000000000000005e-06,
+      "loss": 0.7703,
+      "step": 18
+    },
+    {
+      "epoch": 0.06168831168831169,
+      "grad_norm": 3.90625,
+      "learning_rate": 7.600000000000001e-06,
+      "loss": 0.7327,
+      "step": 19
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "grad_norm": 2.78125,
+      "learning_rate": 8.000000000000001e-06,
+      "loss": 0.6672,
+      "step": 20
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "eval_loss": 0.6521233320236206,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.999,
+      "eval_samples_per_second": 11.88,
+      "eval_steps_per_second": 1.486,
+      "step": 20
+    },
+    {
+      "epoch": 0.06818181818181818,
+      "grad_norm": 2.140625,
+      "learning_rate": 8.400000000000001e-06,
+      "loss": 0.6507,
+      "step": 21
+    },
+    {
+      "epoch": 0.07142857142857142,
+      "grad_norm": 1.828125,
+      "learning_rate": 8.8e-06,
+      "loss": 0.6215,
+      "step": 22
+    },
+    {
+      "epoch": 0.07467532467532467,
+      "grad_norm": 1.5546875,
+      "learning_rate": 9.200000000000002e-06,
+      "loss": 0.5992,
+      "step": 23
+    },
+    {
+      "epoch": 0.07792207792207792,
+      "grad_norm": 1.4453125,
+      "learning_rate": 9.600000000000001e-06,
+      "loss": 0.5748,
+      "step": 24
+    },
+    {
+      "epoch": 0.08116883116883117,
+      "grad_norm": 1.328125,
+      "learning_rate": 1e-05,
+      "loss": 0.5572,
+      "step": 25
+    },
+    {
+      "epoch": 0.08441558441558442,
+      "grad_norm": 1.4765625,
+      "learning_rate": 1.04e-05,
+      "loss": 0.542,
+      "step": 26
+    },
+    {
+      "epoch": 0.08766233766233766,
+      "grad_norm": 1.1171875,
+      "learning_rate": 1.0800000000000002e-05,
+      "loss": 0.5317,
+      "step": 27
+    },
+    {
+      "epoch": 0.09090909090909091,
+      "grad_norm": 0.90625,
+      "learning_rate": 1.1200000000000001e-05,
+      "loss": 0.5173,
+      "step": 28
+    },
+    {
+      "epoch": 0.09415584415584416,
+      "grad_norm": 0.83984375,
+      "learning_rate": 1.16e-05,
+      "loss": 0.4892,
+      "step": 29
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "grad_norm": 0.828125,
+      "learning_rate": 1.2e-05,
+      "loss": 0.491,
+      "step": 30
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "eval_loss": 0.48583006858825684,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9705,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 30
+    },
+    {
+      "epoch": 0.10064935064935066,
+      "grad_norm": 0.8359375,
+      "learning_rate": 1.2400000000000002e-05,
+      "loss": 0.4852,
+      "step": 31
+    },
+    {
+      "epoch": 0.1038961038961039,
+      "grad_norm": 1.0078125,
+      "learning_rate": 1.2800000000000001e-05,
+      "loss": 0.4749,
+      "step": 32
+    },
+    {
+      "epoch": 0.10714285714285714,
+      "grad_norm": 0.99609375,
+      "learning_rate": 1.3200000000000002e-05,
+      "loss": 0.467,
+      "step": 33
+    },
+    {
+      "epoch": 0.11038961038961038,
+      "grad_norm": 0.734375,
+      "learning_rate": 1.3600000000000002e-05,
+      "loss": 0.4545,
+      "step": 34
+    },
+    {
+      "epoch": 0.11363636363636363,
+      "grad_norm": 0.69140625,
+      "learning_rate": 1.4e-05,
+      "loss": 0.4609,
+      "step": 35
+    },
+    {
+      "epoch": 0.11688311688311688,
+      "grad_norm": 0.69921875,
+      "learning_rate": 1.4400000000000001e-05,
+      "loss": 0.4497,
+      "step": 36
+    },
+    {
+      "epoch": 0.12012987012987013,
+      "grad_norm": 0.703125,
+      "learning_rate": 1.48e-05,
+      "loss": 0.4607,
+      "step": 37
+    },
+    {
+      "epoch": 0.12337662337662338,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5200000000000002e-05,
+      "loss": 0.443,
+      "step": 38
+    },
+    {
+      "epoch": 0.1266233766233766,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.5600000000000003e-05,
+      "loss": 0.4428,
+      "step": 39
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.6000000000000003e-05,
+      "loss": 0.435,
+      "step": 40
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "eval_loss": 0.4367639422416687,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0177,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 40
+    },
+    {
+      "epoch": 0.1331168831168831,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.64e-05,
+      "loss": 0.438,
+      "step": 41
+    },
+    {
+      "epoch": 0.13636363636363635,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.6800000000000002e-05,
+      "loss": 0.4392,
+      "step": 42
+    },
+    {
+      "epoch": 0.1396103896103896,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.72e-05,
+      "loss": 0.435,
+      "step": 43
+    },
+    {
+      "epoch": 0.14285714285714285,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.76e-05,
+      "loss": 0.4356,
+      "step": 44
+    },
+    {
+      "epoch": 0.1461038961038961,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.8e-05,
+      "loss": 0.4172,
+      "step": 45
+    },
+    {
+      "epoch": 0.14935064935064934,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.8400000000000003e-05,
+      "loss": 0.4162,
+      "step": 46
+    },
+    {
+      "epoch": 0.1525974025974026,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.88e-05,
+      "loss": 0.418,
+      "step": 47
+    },
+    {
+      "epoch": 0.15584415584415584,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.9200000000000003e-05,
+      "loss": 0.4177,
+      "step": 48
+    },
+    {
+      "epoch": 0.1590909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9600000000000002e-05,
+      "loss": 0.4064,
+      "step": 49
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "grad_norm": 0.62890625,
+      "learning_rate": 2e-05,
+      "loss": 0.4078,
+      "step": 50
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "eval_loss": 0.4077843129634857,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.8482,
+      "eval_samples_per_second": 11.89,
+      "eval_steps_per_second": 1.487,
+      "step": 50
+    },
+    {
+      "epoch": 0.16558441558441558,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9999258647132645e-05,
+      "loss": 0.4057,
+      "step": 51
+    },
+    {
+      "epoch": 0.16883116883116883,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.9997034698451396e-05,
+      "loss": 0.4172,
+      "step": 52
+    },
+    {
+      "epoch": 0.17207792207792208,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9993328483702393e-05,
+      "loss": 0.4082,
+      "step": 53
+    },
+    {
+      "epoch": 0.17532467532467533,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998814055240823e-05,
+      "loss": 0.3973,
+      "step": 54
+    },
+    {
+      "epoch": 0.17857142857142858,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998147167378645e-05,
+      "loss": 0.3969,
+      "step": 55
+    },
+    {
+      "epoch": 0.18181818181818182,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9973322836635517e-05,
+      "loss": 0.4032,
+      "step": 56
+    },
+    {
+      "epoch": 0.18506493506493507,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9963695249188185e-05,
+      "loss": 0.3956,
+      "step": 57
+    },
+    {
+      "epoch": 0.18831168831168832,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.995259033893236e-05,
+      "loss": 0.3952,
+      "step": 58
+    },
+    {
+      "epoch": 0.19155844155844157,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9940009752399462e-05,
+      "loss": 0.3905,
+      "step": 59
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9925955354920265e-05,
+      "loss": 0.3822,
+      "step": 60
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "eval_loss": 0.38575297594070435,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9122,
+      "eval_samples_per_second": 11.886,
+      "eval_steps_per_second": 1.486,
+      "step": 60
+    },
+    {
+      "epoch": 0.19805194805194806,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9910429230348348e-05,
+      "loss": 0.3976,
+      "step": 61
+    },
+    {
+      "epoch": 0.2012987012987013,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9893433680751105e-05,
+      "loss": 0.3714,
+      "step": 62
+    },
+    {
+      "epoch": 0.20454545454545456,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9874971226068417e-05,
+      "loss": 0.3739,
+      "step": 63
+    },
+    {
+      "epoch": 0.2077922077922078,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.985504460373903e-05,
+      "loss": 0.3907,
+      "step": 64
+    },
+    {
+      "epoch": 0.21103896103896103,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.983365676829466e-05,
+      "loss": 0.3803,
+      "step": 65
+    },
+    {
+      "epoch": 0.21428571428571427,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9810810890921943e-05,
+      "loss": 0.3902,
+      "step": 66
+    },
+    {
+      "epoch": 0.21753246753246752,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9786510358992213e-05,
+      "loss": 0.3744,
+      "step": 67
+    },
+    {
+      "epoch": 0.22077922077922077,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.9760758775559275e-05,
+      "loss": 0.3668,
+      "step": 68
+    },
+    {
+      "epoch": 0.22402597402597402,
+      "grad_norm": 0.625,
+      "learning_rate": 1.9733559958825167e-05,
+      "loss": 0.3755,
+      "step": 69
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9704917941574053e-05,
+      "loss": 0.3733,
+      "step": 70
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "eval_loss": 0.3694484233856201,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9363,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 70
+    },
+    {
+      "epoch": 0.2305194805194805,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.9674836970574253e-05,
+      "loss": 0.3676,
+      "step": 71
+    },
+    {
+      "epoch": 0.23376623376623376,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9643321505948588e-05,
+      "loss": 0.3721,
+      "step": 72
+    },
+    {
+      "epoch": 0.237012987012987,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9610376220513067e-05,
+      "loss": 0.3558,
+      "step": 73
+    },
+    {
+      "epoch": 0.24025974025974026,
+      "grad_norm": 0.625,
+      "learning_rate": 1.957600599908406e-05,
+      "loss": 0.357,
+      "step": 74
+    },
+    {
+      "epoch": 0.2435064935064935,
+      "grad_norm": 0.66015625,
+      "learning_rate": 1.954021593775401e-05,
+      "loss": 0.3608,
+      "step": 75
+    },
+    {
+      "epoch": 0.24675324675324675,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9503011343135828e-05,
+      "loss": 0.3513,
+      "step": 76
+    },
+    {
+      "epoch": 0.25,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9464397731576093e-05,
+      "loss": 0.3693,
+      "step": 77
+    },
+    {
+      "epoch": 0.2532467532467532,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9424380828337146e-05,
+      "loss": 0.3605,
+      "step": 78
+    },
+    {
+      "epoch": 0.2564935064935065,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.938296656674817e-05,
+      "loss": 0.3553,
+      "step": 79
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.9340161087325483e-05,
+      "loss": 0.3721,
+      "step": 80
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "eval_loss": 0.3553767800331116,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0277,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 80
+    },
+    {
+      "epoch": 0.262987012987013,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9295970736862063e-05,
+      "loss": 0.3597,
+      "step": 81
+    },
+    {
+      "epoch": 0.2662337662337662,
+      "grad_norm": 0.65625,
+      "learning_rate": 1.9250402067486523e-05,
+      "loss": 0.3527,
+      "step": 82
+    },
+    {
+      "epoch": 0.2694805194805195,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9203461835691596e-05,
+      "loss": 0.3528,
+      "step": 83
+    },
+    {
+      "epoch": 0.2727272727272727,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.9155157001332374e-05,
+      "loss": 0.356,
+      "step": 84
+    },
+    {
+      "epoch": 0.275974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9105494726594344e-05,
+      "loss": 0.3556,
+      "step": 85
+    },
+    {
+      "epoch": 0.2792207792207792,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.905448237493147e-05,
+      "loss": 0.3473,
+      "step": 86
+    },
+    {
+      "epoch": 0.2824675324675325,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9002127509974376e-05,
+      "loss": 0.3441,
+      "step": 87
+    },
+    {
+      "epoch": 0.2857142857142857,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.894843789440892e-05,
+      "loss": 0.3492,
+      "step": 88
+    },
+    {
+      "epoch": 0.288961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.889342148882519e-05,
+      "loss": 0.3338,
+      "step": 89
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "grad_norm": 0.67578125,
+      "learning_rate": 1.8837086450537195e-05,
+      "loss": 0.3347,
+      "step": 90
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "eval_loss": 0.34332525730133057,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9257,
+      "eval_samples_per_second": 11.885,
+      "eval_steps_per_second": 1.486,
+      "step": 90
+    },
+    {
+      "epoch": 0.29545454545454547,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.877944113237336e-05,
+      "loss": 0.3438,
+      "step": 91
+    },
+    {
+      "epoch": 0.2987012987012987,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.872049408143808e-05,
+      "loss": 0.3505,
+      "step": 92
+    },
+    {
+      "epoch": 0.30194805194805197,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.866025403784439e-05,
+      "loss": 0.3362,
+      "step": 93
+    },
+    {
+      "epoch": 0.3051948051948052,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.8598729933418102e-05,
+      "loss": 0.337,
+      "step": 94
+    },
+    {
+      "epoch": 0.30844155844155846,
+      "grad_norm": 0.64453125,
+      "learning_rate": 1.8535930890373467e-05,
+      "loss": 0.3324,
+      "step": 95
+    },
+    {
+      "epoch": 0.3116883116883117,
+      "grad_norm": 0.6875,
+      "learning_rate": 1.8471866219960604e-05,
+      "loss": 0.3461,
+      "step": 96
+    },
+    {
+      "epoch": 0.31493506493506496,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.840654542108494e-05,
+      "loss": 0.3318,
+      "step": 97
+    },
+    {
+      "epoch": 0.3181818181818182,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.833997817889878e-05,
+      "loss": 0.3335,
+      "step": 98
+    },
+    {
+      "epoch": 0.32142857142857145,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.82721743633653e-05,
+      "loss": 0.3291,
+      "step": 99
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.820314402779511e-05,
+      "loss": 0.3324,
+      "step": 100
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "eval_loss": 0.33332687616348267,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9506,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 100
+    },
+    {
+      "epoch": 0.32792207792207795,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.8132897407355657e-05,
+      "loss": 0.3288,
+      "step": 101
+    },
+    {
+      "epoch": 0.33116883116883117,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.806144491755363e-05,
+      "loss": 0.3384,
+      "step": 102
+    },
+    {
+      "epoch": 0.3344155844155844,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.798879715269067e-05,
+      "loss": 0.3329,
+      "step": 103
+    },
+    {
+      "epoch": 0.33766233766233766,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.7914964884292543e-05,
+      "loss": 0.3329,
+      "step": 104
+    },
+    {
+      "epoch": 0.3409090909090909,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.7839959059512016e-05,
+      "loss": 0.3335,
+      "step": 105
+    },
+    {
+      "epoch": 0.34415584415584416,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.7763790799505746e-05,
+      "loss": 0.3201,
+      "step": 106
+    },
+    {
+      "epoch": 0.3474025974025974,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.7686471397785322e-05,
+      "loss": 0.3327,
+      "step": 107
+    },
+    {
+      "epoch": 0.35064935064935066,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.760801231854278e-05,
+      "loss": 0.3205,
+      "step": 108
+    },
+    {
+      "epoch": 0.3538961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.7528425194950794e-05,
+      "loss": 0.334,
+      "step": 109
+    },
+    {
+      "epoch": 0.35714285714285715,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.744772182743782e-05,
+      "loss": 0.3165,
+      "step": 110
+    },
+    {
+      "epoch": 0.35714285714285715,
+      "eval_loss": 0.32505103945732117,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0106,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 110
+    },
+    {
+      "epoch": 0.36038961038961037,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.736591418193844e-05,
+      "loss": 0.3236,
+      "step": 111
+    },
+    {
+      "epoch": 0.36363636363636365,
+      "grad_norm": 0.63671875,
+      "learning_rate": 1.728301438811916e-05,
+      "loss": 0.3192,
+      "step": 112
+    },
+    {
+      "epoch": 0.36688311688311687,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.7199034737579962e-05,
+      "loss": 0.327,
+      "step": 113
+    },
+    {
+      "epoch": 0.37012987012987014,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.711398768203178e-05,
+      "loss": 0.3169,
+      "step": 114
+    },
+    {
+      "epoch": 0.37337662337662336,
+      "grad_norm": 0.625,
+      "learning_rate": 1.7027885831450318e-05,
+      "loss": 0.3186,
+      "step": 115
+    },
+    {
+      "epoch": 0.37662337662337664,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.6940741952206342e-05,
+      "loss": 0.3136,
+      "step": 116
+    },
+    {
+      "epoch": 0.37987012987012986,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.6852568965172794e-05,
+      "loss": 0.3184,
+      "step": 117
+    },
+    {
+      "epoch": 0.38311688311688313,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.676337994380903e-05,
+      "loss": 0.3114,
+      "step": 118
+    },
+    {
+      "epoch": 0.38636363636363635,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.6673188112222394e-05,
+      "loss": 0.3329,
+      "step": 119
+    },
+    {
+      "epoch": 0.38961038961038963,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.658200684320748e-05,
+      "loss": 0.3105,
+      "step": 120
+    },
+    {
+      "epoch": 0.38961038961038963,
+      "eval_loss": 0.31796982884407043,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9896,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 120
+    },
+    {
+      "epoch": 0.39285714285714285,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.6489849656263336e-05,
+      "loss": 0.3084,
+      "step": 121
+    },
+    {
+      "epoch": 0.3961038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.6396730215588913e-05,
+      "loss": 0.3124,
+      "step": 122
+    },
+    {
+      "epoch": 0.39935064935064934,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.630266232805709e-05,
+      "loss": 0.3167,
+      "step": 123
+    },
+    {
+      "epoch": 0.4025974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.6207659941167485e-05,
+      "loss": 0.3084,
+      "step": 124
+    },
+    {
+      "epoch": 0.40584415584415584,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.6111737140978495e-05,
+      "loss": 0.3143,
+      "step": 125
+    },
+    {
+      "epoch": 0.4090909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.6014908150018703e-05,
+      "loss": 0.3173,
+      "step": 126
+    },
+    {
+      "epoch": 0.41233766233766234,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.591718732517814e-05,
+      "loss": 0.3199,
+      "step": 127
+    },
+    {
+      "epoch": 0.4155844155844156,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.581858915557953e-05,
+      "loss": 0.312,
+      "step": 128
+    },
+    {
+      "epoch": 0.41883116883116883,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.571912826043003e-05,
+      "loss": 0.3217,
+      "step": 129
+    },
+    {
+      "epoch": 0.42207792207792205,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.5618819386853607e-05,
+      "loss": 0.3123,
+      "step": 130
+    },
+    {
+      "epoch": 0.42207792207792205,
+      "eval_loss": 0.3119741976261139,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9669,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 130
+    },
+    {
+      "epoch": 0.4253246753246753,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.551767740770446e-05,
+      "loss": 0.3088,
+      "step": 131
+    },
+    {
+      "epoch": 0.42857142857142855,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.541571731936185e-05,
+      "loss": 0.3084,
+      "step": 132
+    },
+    {
+      "epoch": 0.4318181818181818,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5312954239506536e-05,
+      "loss": 0.3179,
+      "step": 133
+    },
+    {
+      "epoch": 0.43506493506493504,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.5209403404879305e-05,
+      "loss": 0.3145,
+      "step": 134
+    },
+    {
+      "epoch": 0.4383116883116883,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.5105080169021792e-05,
+      "loss": 0.3045,
+      "step": 135
+    },
+    {
+      "epoch": 0.44155844155844154,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.5000000000000002e-05,
+      "loss": 0.314,
+      "step": 136
+    },
+    {
+      "epoch": 0.4448051948051948,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.4894178478110856e-05,
+      "loss": 0.2985,
+      "step": 137
+    },
+    {
+      "epoch": 0.44805194805194803,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.4787631293572094e-05,
+      "loss": 0.3114,
+      "step": 138
+    },
+    {
+      "epoch": 0.4512987012987013,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.4680374244195861e-05,
+      "loss": 0.3129,
+      "step": 139
+    },
+    {
+      "epoch": 0.45454545454545453,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.4572423233046386e-05,
+      "loss": 0.3007,
+      "step": 140
+    },
+    {
+      "epoch": 0.45454545454545453,
+      "eval_loss": 0.30710509419441223,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9622,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 140
+    },
+    {
+      "epoch": 0.4577922077922078,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.4463794266081994e-05,
+      "loss": 0.2981,
+      "step": 141
+    },
+    {
+      "epoch": 0.461038961038961,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.4354503449781914e-05,
+      "loss": 0.2974,
+      "step": 142
+    },
+    {
+      "epoch": 0.4642857142857143,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.4244566988758152e-05,
+      "loss": 0.3064,
+      "step": 143
+    },
+    {
+      "epoch": 0.4675324675324675,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.4134001183352833e-05,
+      "loss": 0.298,
+      "step": 144
+    },
+    {
+      "epoch": 0.4707792207792208,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.4022822427221325e-05,
+      "loss": 0.3037,
+      "step": 145
+    },
+    {
+      "epoch": 0.474025974025974,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.391104720490156e-05,
+      "loss": 0.3062,
+      "step": 146
+    },
+    {
+      "epoch": 0.4772727272727273,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.3798692089369855e-05,
+      "loss": 0.3019,
+      "step": 147
+    },
+    {
+      "epoch": 0.4805194805194805,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.368577373958362e-05,
+      "loss": 0.3008,
+      "step": 148
+    },
+    {
+      "epoch": 0.4837662337662338,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.3572308898011328e-05,
+      "loss": 0.3055,
+      "step": 149
+    },
+    {
+      "epoch": 0.487012987012987,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.3458314388150115e-05,
+      "loss": 0.2942,
+      "step": 150
+    },
+    {
+      "epoch": 0.487012987012987,
+      "eval_loss": 0.30315953493118286,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9566,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 150
+    },
+    {
+      "epoch": 0.4902597402597403,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.3343807112031329e-05,
+      "loss": 0.3097,
+      "step": 151
+    },
+    {
+      "epoch": 0.4935064935064935,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.3228804047714462e-05,
+      "loss": 0.3004,
+      "step": 152
+    },
+    {
+      "epoch": 0.4967532467532468,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.3113322246769817e-05,
+      "loss": 0.3046,
+      "step": 153
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.2997378831750242e-05,
+      "loss": 0.2966,
+      "step": 154
+    },
+    {
+      "epoch": 0.5032467532467533,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.2880990993652379e-05,
+      "loss": 0.2989,
+      "step": 155
+    },
+    {
+      "epoch": 0.5064935064935064,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.2764175989367717e-05,
+      "loss": 0.3015,
+      "step": 156
+    },
+    {
+      "epoch": 0.5097402597402597,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.2646951139123935e-05,
+      "loss": 0.2976,
+      "step": 157
+    },
+    {
+      "epoch": 0.512987012987013,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.2529333823916807e-05,
+      "loss": 0.2919,
+      "step": 158
+    },
+    {
+      "epoch": 0.5162337662337663,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.241134148293311e-05,
+      "loss": 0.3041,
+      "step": 159
+    },
+    {
+      "epoch": 0.5194805194805194,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.2292991610964902e-05,
+      "loss": 0.3053,
+      "step": 160
+    },
+    {
+      "epoch": 0.5194805194805194,
+      "eval_loss": 0.3000594675540924,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9643,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 160
+    },
+    {
+      "epoch": 0.5227272727272727,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.2174301755815572e-05,
+      "loss": 0.3043,
+      "step": 161
+    },
+    {
+      "epoch": 0.525974025974026,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.2055289515698008e-05,
+      "loss": 0.2988,
+      "step": 162
+    },
+    {
+      "epoch": 0.5292207792207793,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1935972536625302e-05,
+      "loss": 0.2871,
+      "step": 163
+    },
+    {
+      "epoch": 0.5324675324675324,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.1816368509794365e-05,
+      "loss": 0.2985,
+      "step": 164
+    },
+    {
+      "epoch": 0.5357142857142857,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.1696495168962848e-05,
+      "loss": 0.3066,
+      "step": 165
+    },
+    {
+      "epoch": 0.538961038961039,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.1576370287819737e-05,
+      "loss": 0.296,
+      "step": 166
+    },
+    {
+      "epoch": 0.5422077922077922,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.1456011677350052e-05,
+      "loss": 0.2997,
+      "step": 167
+    },
+    {
+      "epoch": 0.5454545454545454,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.133543718319398e-05,
+      "loss": 0.3011,
+      "step": 168
+    },
+    {
+      "epoch": 0.5487012987012987,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1214664683000927e-05,
+      "loss": 0.2898,
+      "step": 169
+    },
+    {
+      "epoch": 0.551948051948052,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.1093712083778748e-05,
+      "loss": 0.292,
+      "step": 170
+    },
+    {
+      "epoch": 0.551948051948052,
+      "eval_loss": 0.29717138409614563,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9869,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 170
+    },
+    {
+      "epoch": 0.5551948051948052,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.0972597319238692e-05,
+      "loss": 0.2995,
+      "step": 171
+    },
+    {
+      "epoch": 0.5584415584415584,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.0851338347136358e-05,
+      "loss": 0.3015,
+      "step": 172
+    },
+    {
+      "epoch": 0.5616883116883117,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.0729953146609076e-05,
+      "loss": 0.2953,
+      "step": 173
+    },
+    {
+      "epoch": 0.564935064935065,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.060845971551014e-05,
+      "loss": 0.3011,
+      "step": 174
+    },
+    {
+      "epoch": 0.5681818181818182,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.0486876067740253e-05,
+      "loss": 0.3016,
+      "step": 175
+    },
+    {
+      "epoch": 0.5714285714285714,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.0365220230576592e-05,
+      "loss": 0.2953,
+      "step": 176
+    },
+    {
+      "epoch": 0.5746753246753247,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.0243510241999898e-05,
+      "loss": 0.2948,
+      "step": 177
+    },
+    {
+      "epoch": 0.577922077922078,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.0121764148019977e-05,
+      "loss": 0.2958,
+      "step": 178
+    },
+    {
+      "epoch": 0.5811688311688312,
+      "grad_norm": 0.5546875,
+      "learning_rate": 1e-05,
+      "loss": 0.2829,
+      "step": 179
+    },
+    {
+      "epoch": 0.5844155844155844,
+      "grad_norm": 0.58203125,
+      "learning_rate": 9.878235851980027e-06,
+      "loss": 0.2917,
+      "step": 180
+    },
+    {
+      "epoch": 0.5844155844155844,
+      "eval_loss": 0.29469791054725647,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9839,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 180
+    },
+    {
+      "epoch": 0.5876623376623377,
+      "grad_norm": 0.59375,
+      "learning_rate": 9.756489758000105e-06,
+      "loss": 0.2994,
+      "step": 181
+    },
+    {
+      "epoch": 0.5909090909090909,
+      "grad_norm": 0.57421875,
+      "learning_rate": 9.634779769423412e-06,
+      "loss": 0.3054,
+      "step": 182
+    },
+    {
+      "epoch": 0.5941558441558441,
+      "grad_norm": 0.59375,
+      "learning_rate": 9.51312393225975e-06,
+      "loss": 0.2925,
+      "step": 183
+    },
+    {
+      "epoch": 0.5974025974025974,
+      "grad_norm": 0.5703125,
+      "learning_rate": 9.391540284489862e-06,
+      "loss": 0.2816,
+      "step": 184
+    },
+    {
+      "epoch": 0.6006493506493507,
+      "grad_norm": 0.5390625,
+      "learning_rate": 9.270046853390924e-06,
+      "loss": 0.2856,
+      "step": 185
+    },
+    {
+      "epoch": 0.6038961038961039,
+      "grad_norm": 0.5703125,
+      "learning_rate": 9.148661652863644e-06,
+      "loss": 0.2936,
+      "step": 186
+    },
+    {
+      "epoch": 0.6071428571428571,
+      "grad_norm": 0.55078125,
+      "learning_rate": 9.02740268076131e-06,
+      "loss": 0.2938,
+      "step": 187
+    },
+    {
+      "epoch": 0.6103896103896104,
+      "grad_norm": 0.5859375,
+      "learning_rate": 8.906287916221259e-06,
+      "loss": 0.3017,
+      "step": 188
+    },
+    {
+      "epoch": 0.6136363636363636,
+      "grad_norm": 0.578125,
+      "learning_rate": 8.785335316999078e-06,
+      "loss": 0.2957,
+      "step": 189
+    },
+    {
+      "epoch": 0.6168831168831169,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.664562816806022e-06,
+      "loss": 0.2961,
+      "step": 190
+    },
+    {
+      "epoch": 0.6168831168831169,
+      "eval_loss": 0.2927069365978241,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9405,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 190
+    },
+    {
+      "epoch": 0.6201298701298701,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.543988322649954e-06,
+      "loss": 0.2915,
+      "step": 191
+    },
+    {
+      "epoch": 0.6233766233766234,
+      "grad_norm": 0.57421875,
+      "learning_rate": 8.423629712180265e-06,
+      "loss": 0.2875,
+      "step": 192
+    },
+    {
+      "epoch": 0.6266233766233766,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.303504831037154e-06,
+      "loss": 0.2979,
+      "step": 193
+    },
+    {
+      "epoch": 0.6298701298701299,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.183631490205636e-06,
+      "loss": 0.2891,
+      "step": 194
+    },
+    {
+      "epoch": 0.6331168831168831,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.064027463374702e-06,
+      "loss": 0.2796,
+      "step": 195
+    },
+    {
+      "epoch": 0.6363636363636364,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.944710484301995e-06,
+      "loss": 0.285,
+      "step": 196
+    },
+    {
+      "epoch": 0.6396103896103896,
+      "grad_norm": 0.58203125,
+      "learning_rate": 7.825698244184432e-06,
+      "loss": 0.295,
+      "step": 197
+    },
+    {
+      "epoch": 0.6428571428571429,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.707008389035102e-06,
+      "loss": 0.2962,
+      "step": 198
+    },
+    {
+      "epoch": 0.6461038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.588658517066893e-06,
+      "loss": 0.2912,
+      "step": 199
+    },
+    {
+      "epoch": 0.6493506493506493,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.470666176083193e-06,
+      "loss": 0.2917,
+      "step": 200
+    },
+    {
+      "epoch": 0.6493506493506493,
+      "eval_loss": 0.2912693917751312,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9875,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 200
+    }
+  ],
+  "logging_steps": 1,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 100,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1.1804225839300608e+18,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/checkpoint-200/training_args.bin b/checkpoint-200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..4684d3c90558d6bfde405c82b58596bc6231bfee
--- /dev/null
+++ b/checkpoint-200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d09f69be12f7ef607f0a1508435a517f5277e3160e5d6a59346e619fcbaa3592
+size 5176
diff --git a/checkpoint-300/config.json b/checkpoint-300/config.json
new file mode 100644
index 0000000000000000000000000000000000000000..de55730a672c2805e1108109d81a111b0cdb48b1
--- /dev/null
+++ b/checkpoint-300/config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f8b621112eddc9da772fae3f70d672f4d949cc5cb1d580f135d66b3836598c1a
+size 915
diff --git a/checkpoint-300/generation_config.json b/checkpoint-300/generation_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..1007819213eaa3e36dd0c6354016734419f4a863
--- /dev/null
+++ b/checkpoint-300/generation_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a7327788370b7783e29852cf19b4a1f40cd31af5962c94bba76c336d491b4169
+size 184
diff --git a/checkpoint-300/model-00001-of-00004.safetensors b/checkpoint-300/model-00001-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..16a7cb28d9d5a8bc030ff58aa7bb2aec9c884506
--- /dev/null
+++ b/checkpoint-300/model-00001-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:29a3907a94d16ee63bb35418803792e2e7faf7479e9a0ba6c5b6faf863b4007d
+size 4976698672
diff --git a/checkpoint-300/model-00002-of-00004.safetensors b/checkpoint-300/model-00002-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..ea6d56cc2a5e3faa773bd777c6c8d20e58191164
--- /dev/null
+++ b/checkpoint-300/model-00002-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2ed5d4c274155ab7efb36612cbbb7e548c1b6b6fdb0e8ef136061e0c5b26b129
+size 4999802720
diff --git a/checkpoint-300/model-00003-of-00004.safetensors b/checkpoint-300/model-00003-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..0ee4f7a839f73a703690a98c80e9ff4f588dd775
--- /dev/null
+++ b/checkpoint-300/model-00003-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b06c1356e06091e7d05ca40b62cd7a438c1abb71c6e182bd479804b9bf04e4e2
+size 4915916176
diff --git a/checkpoint-300/model-00004-of-00004.safetensors b/checkpoint-300/model-00004-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..c24ce7664aca02faaeb680746c2c99eec37a0249
--- /dev/null
+++ b/checkpoint-300/model-00004-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:eeb6b9ac5806653b828e732fba12471ef3f300d4d9daa99da11f1cf958f76402
+size 1168138808
diff --git a/checkpoint-300/model.safetensors.index.json b/checkpoint-300/model.safetensors.index.json
new file mode 100644
index 0000000000000000000000000000000000000000..a054aad8cf0cb5671e2e10b3bf817725a4bec031
--- /dev/null
+++ b/checkpoint-300/model.safetensors.index.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:146776fce3f6db1103aa6f249e65ee5544c5923ce6f971b092eee79aa6e5d37b
+size 23950
diff --git a/checkpoint-300/optimizer.pt b/checkpoint-300/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..5e5684dac752410d520d02408132a8c1a96f699d
--- /dev/null
+++ b/checkpoint-300/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ee6d30131aec5e4206a28344f043ab4470213fc7de969421738254b908eb6f93
+size 32121299754
diff --git a/checkpoint-300/rng_state.pth b/checkpoint-300/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..02e6a75cccce6d8b138f83bf61f04a2d82786ce4
--- /dev/null
+++ b/checkpoint-300/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:374644995be79c967030295cf4e5929f6a33d3b1197b8855bc39718b44686552
+size 14244
diff --git a/checkpoint-300/scheduler.pt b/checkpoint-300/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..6726f19244a9d9c1cc38695d2f8e7194d58f3aa2
--- /dev/null
+++ b/checkpoint-300/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:892afc48e852cff639a2b1d3faa9e143f11c0af53826ffadf88b7d4e636251cb
+size 1064
diff --git a/checkpoint-300/special_tokens_map.json b/checkpoint-300/special_tokens_map.json
new file mode 100644
index 0000000000000000000000000000000000000000..92d7fab6729d723f45689803b99006fec945cc47
--- /dev/null
+++ b/checkpoint-300/special_tokens_map.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b1835caa5b4d70acaa210fa222b0036f1882f9525c4660fd4810fb3e1e40ff8
+size 325
diff --git a/checkpoint-300/tokenizer.json b/checkpoint-300/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..66cd9d7e0daec95eb10d16a63c615637dbbb7304
--- /dev/null
+++ b/checkpoint-300/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79e3e522635f3171300913bb421464a87de6222182a0570b9b2ccba2a964b2b4
+size 9085657
diff --git a/checkpoint-300/tokenizer_config.json b/checkpoint-300/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..08ebc9598183a9a76c44ef06813aca75f51de4a5
--- /dev/null
+++ b/checkpoint-300/tokenizer_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:05d0de6d831b5940c3f2eae4e495b9f77612cbb05b5804846733ab617f0dfe9b
+size 55380
diff --git a/checkpoint-300/trainer_state.json b/checkpoint-300/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c54de5afde22ae57b327ae39425beaca4aaffb17
--- /dev/null
+++ b/checkpoint-300/trainer_state.json
@@ -0,0 +1,2403 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.974025974025974,
+  "eval_steps": 10,
+  "global_step": 300,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.003246753246753247,
+      "grad_norm": 16.375,
+      "learning_rate": 4.0000000000000003e-07,
+      "loss": 1.4017,
+      "step": 1
+    },
+    {
+      "epoch": 0.006493506493506494,
+      "grad_norm": 16.25,
+      "learning_rate": 8.000000000000001e-07,
+      "loss": 1.4229,
+      "step": 2
+    },
+    {
+      "epoch": 0.00974025974025974,
+      "grad_norm": 16.75,
+      "learning_rate": 1.2000000000000002e-06,
+      "loss": 1.4112,
+      "step": 3
+    },
+    {
+      "epoch": 0.012987012987012988,
+      "grad_norm": 16.5,
+      "learning_rate": 1.6000000000000001e-06,
+      "loss": 1.3959,
+      "step": 4
+    },
+    {
+      "epoch": 0.016233766233766232,
+      "grad_norm": 16.125,
+      "learning_rate": 2.0000000000000003e-06,
+      "loss": 1.3789,
+      "step": 5
+    },
+    {
+      "epoch": 0.01948051948051948,
+      "grad_norm": 14.4375,
+      "learning_rate": 2.4000000000000003e-06,
+      "loss": 1.3618,
+      "step": 6
+    },
+    {
+      "epoch": 0.022727272727272728,
+      "grad_norm": 10.1875,
+      "learning_rate": 2.8000000000000003e-06,
+      "loss": 1.3214,
+      "step": 7
+    },
+    {
+      "epoch": 0.025974025974025976,
+      "grad_norm": 7.4375,
+      "learning_rate": 3.2000000000000003e-06,
+      "loss": 1.2885,
+      "step": 8
+    },
+    {
+      "epoch": 0.02922077922077922,
+      "grad_norm": 6.28125,
+      "learning_rate": 3.6000000000000003e-06,
+      "loss": 1.2436,
+      "step": 9
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "grad_norm": 5.75,
+      "learning_rate": 4.000000000000001e-06,
+      "loss": 1.2269,
+      "step": 10
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "eval_loss": 1.1976113319396973,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0273,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 10
+    },
+    {
+      "epoch": 0.03571428571428571,
+      "grad_norm": 5.375,
+      "learning_rate": 4.4e-06,
+      "loss": 1.1925,
+      "step": 11
+    },
+    {
+      "epoch": 0.03896103896103896,
+      "grad_norm": 5.125,
+      "learning_rate": 4.800000000000001e-06,
+      "loss": 1.1307,
+      "step": 12
+    },
+    {
+      "epoch": 0.04220779220779221,
+      "grad_norm": 5.125,
+      "learning_rate": 5.2e-06,
+      "loss": 1.0778,
+      "step": 13
+    },
+    {
+      "epoch": 0.045454545454545456,
+      "grad_norm": 5.0625,
+      "learning_rate": 5.600000000000001e-06,
+      "loss": 1.0067,
+      "step": 14
+    },
+    {
+      "epoch": 0.048701298701298704,
+      "grad_norm": 5.75,
+      "learning_rate": 6e-06,
+      "loss": 0.9668,
+      "step": 15
+    },
+    {
+      "epoch": 0.05194805194805195,
+      "grad_norm": 5.15625,
+      "learning_rate": 6.4000000000000006e-06,
+      "loss": 0.875,
+      "step": 16
+    },
+    {
+      "epoch": 0.05519480519480519,
+      "grad_norm": 4.875,
+      "learning_rate": 6.800000000000001e-06,
+      "loss": 0.8297,
+      "step": 17
+    },
+    {
+      "epoch": 0.05844155844155844,
+      "grad_norm": 4.5625,
+      "learning_rate": 7.2000000000000005e-06,
+      "loss": 0.7703,
+      "step": 18
+    },
+    {
+      "epoch": 0.06168831168831169,
+      "grad_norm": 3.90625,
+      "learning_rate": 7.600000000000001e-06,
+      "loss": 0.7327,
+      "step": 19
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "grad_norm": 2.78125,
+      "learning_rate": 8.000000000000001e-06,
+      "loss": 0.6672,
+      "step": 20
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "eval_loss": 0.6521233320236206,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.999,
+      "eval_samples_per_second": 11.88,
+      "eval_steps_per_second": 1.486,
+      "step": 20
+    },
+    {
+      "epoch": 0.06818181818181818,
+      "grad_norm": 2.140625,
+      "learning_rate": 8.400000000000001e-06,
+      "loss": 0.6507,
+      "step": 21
+    },
+    {
+      "epoch": 0.07142857142857142,
+      "grad_norm": 1.828125,
+      "learning_rate": 8.8e-06,
+      "loss": 0.6215,
+      "step": 22
+    },
+    {
+      "epoch": 0.07467532467532467,
+      "grad_norm": 1.5546875,
+      "learning_rate": 9.200000000000002e-06,
+      "loss": 0.5992,
+      "step": 23
+    },
+    {
+      "epoch": 0.07792207792207792,
+      "grad_norm": 1.4453125,
+      "learning_rate": 9.600000000000001e-06,
+      "loss": 0.5748,
+      "step": 24
+    },
+    {
+      "epoch": 0.08116883116883117,
+      "grad_norm": 1.328125,
+      "learning_rate": 1e-05,
+      "loss": 0.5572,
+      "step": 25
+    },
+    {
+      "epoch": 0.08441558441558442,
+      "grad_norm": 1.4765625,
+      "learning_rate": 1.04e-05,
+      "loss": 0.542,
+      "step": 26
+    },
+    {
+      "epoch": 0.08766233766233766,
+      "grad_norm": 1.1171875,
+      "learning_rate": 1.0800000000000002e-05,
+      "loss": 0.5317,
+      "step": 27
+    },
+    {
+      "epoch": 0.09090909090909091,
+      "grad_norm": 0.90625,
+      "learning_rate": 1.1200000000000001e-05,
+      "loss": 0.5173,
+      "step": 28
+    },
+    {
+      "epoch": 0.09415584415584416,
+      "grad_norm": 0.83984375,
+      "learning_rate": 1.16e-05,
+      "loss": 0.4892,
+      "step": 29
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "grad_norm": 0.828125,
+      "learning_rate": 1.2e-05,
+      "loss": 0.491,
+      "step": 30
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "eval_loss": 0.48583006858825684,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9705,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 30
+    },
+    {
+      "epoch": 0.10064935064935066,
+      "grad_norm": 0.8359375,
+      "learning_rate": 1.2400000000000002e-05,
+      "loss": 0.4852,
+      "step": 31
+    },
+    {
+      "epoch": 0.1038961038961039,
+      "grad_norm": 1.0078125,
+      "learning_rate": 1.2800000000000001e-05,
+      "loss": 0.4749,
+      "step": 32
+    },
+    {
+      "epoch": 0.10714285714285714,
+      "grad_norm": 0.99609375,
+      "learning_rate": 1.3200000000000002e-05,
+      "loss": 0.467,
+      "step": 33
+    },
+    {
+      "epoch": 0.11038961038961038,
+      "grad_norm": 0.734375,
+      "learning_rate": 1.3600000000000002e-05,
+      "loss": 0.4545,
+      "step": 34
+    },
+    {
+      "epoch": 0.11363636363636363,
+      "grad_norm": 0.69140625,
+      "learning_rate": 1.4e-05,
+      "loss": 0.4609,
+      "step": 35
+    },
+    {
+      "epoch": 0.11688311688311688,
+      "grad_norm": 0.69921875,
+      "learning_rate": 1.4400000000000001e-05,
+      "loss": 0.4497,
+      "step": 36
+    },
+    {
+      "epoch": 0.12012987012987013,
+      "grad_norm": 0.703125,
+      "learning_rate": 1.48e-05,
+      "loss": 0.4607,
+      "step": 37
+    },
+    {
+      "epoch": 0.12337662337662338,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5200000000000002e-05,
+      "loss": 0.443,
+      "step": 38
+    },
+    {
+      "epoch": 0.1266233766233766,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.5600000000000003e-05,
+      "loss": 0.4428,
+      "step": 39
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.6000000000000003e-05,
+      "loss": 0.435,
+      "step": 40
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "eval_loss": 0.4367639422416687,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0177,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 40
+    },
+    {
+      "epoch": 0.1331168831168831,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.64e-05,
+      "loss": 0.438,
+      "step": 41
+    },
+    {
+      "epoch": 0.13636363636363635,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.6800000000000002e-05,
+      "loss": 0.4392,
+      "step": 42
+    },
+    {
+      "epoch": 0.1396103896103896,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.72e-05,
+      "loss": 0.435,
+      "step": 43
+    },
+    {
+      "epoch": 0.14285714285714285,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.76e-05,
+      "loss": 0.4356,
+      "step": 44
+    },
+    {
+      "epoch": 0.1461038961038961,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.8e-05,
+      "loss": 0.4172,
+      "step": 45
+    },
+    {
+      "epoch": 0.14935064935064934,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.8400000000000003e-05,
+      "loss": 0.4162,
+      "step": 46
+    },
+    {
+      "epoch": 0.1525974025974026,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.88e-05,
+      "loss": 0.418,
+      "step": 47
+    },
+    {
+      "epoch": 0.15584415584415584,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.9200000000000003e-05,
+      "loss": 0.4177,
+      "step": 48
+    },
+    {
+      "epoch": 0.1590909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9600000000000002e-05,
+      "loss": 0.4064,
+      "step": 49
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "grad_norm": 0.62890625,
+      "learning_rate": 2e-05,
+      "loss": 0.4078,
+      "step": 50
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "eval_loss": 0.4077843129634857,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.8482,
+      "eval_samples_per_second": 11.89,
+      "eval_steps_per_second": 1.487,
+      "step": 50
+    },
+    {
+      "epoch": 0.16558441558441558,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9999258647132645e-05,
+      "loss": 0.4057,
+      "step": 51
+    },
+    {
+      "epoch": 0.16883116883116883,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.9997034698451396e-05,
+      "loss": 0.4172,
+      "step": 52
+    },
+    {
+      "epoch": 0.17207792207792208,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9993328483702393e-05,
+      "loss": 0.4082,
+      "step": 53
+    },
+    {
+      "epoch": 0.17532467532467533,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998814055240823e-05,
+      "loss": 0.3973,
+      "step": 54
+    },
+    {
+      "epoch": 0.17857142857142858,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998147167378645e-05,
+      "loss": 0.3969,
+      "step": 55
+    },
+    {
+      "epoch": 0.18181818181818182,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9973322836635517e-05,
+      "loss": 0.4032,
+      "step": 56
+    },
+    {
+      "epoch": 0.18506493506493507,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9963695249188185e-05,
+      "loss": 0.3956,
+      "step": 57
+    },
+    {
+      "epoch": 0.18831168831168832,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.995259033893236e-05,
+      "loss": 0.3952,
+      "step": 58
+    },
+    {
+      "epoch": 0.19155844155844157,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9940009752399462e-05,
+      "loss": 0.3905,
+      "step": 59
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9925955354920265e-05,
+      "loss": 0.3822,
+      "step": 60
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "eval_loss": 0.38575297594070435,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9122,
+      "eval_samples_per_second": 11.886,
+      "eval_steps_per_second": 1.486,
+      "step": 60
+    },
+    {
+      "epoch": 0.19805194805194806,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9910429230348348e-05,
+      "loss": 0.3976,
+      "step": 61
+    },
+    {
+      "epoch": 0.2012987012987013,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9893433680751105e-05,
+      "loss": 0.3714,
+      "step": 62
+    },
+    {
+      "epoch": 0.20454545454545456,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9874971226068417e-05,
+      "loss": 0.3739,
+      "step": 63
+    },
+    {
+      "epoch": 0.2077922077922078,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.985504460373903e-05,
+      "loss": 0.3907,
+      "step": 64
+    },
+    {
+      "epoch": 0.21103896103896103,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.983365676829466e-05,
+      "loss": 0.3803,
+      "step": 65
+    },
+    {
+      "epoch": 0.21428571428571427,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9810810890921943e-05,
+      "loss": 0.3902,
+      "step": 66
+    },
+    {
+      "epoch": 0.21753246753246752,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9786510358992213e-05,
+      "loss": 0.3744,
+      "step": 67
+    },
+    {
+      "epoch": 0.22077922077922077,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.9760758775559275e-05,
+      "loss": 0.3668,
+      "step": 68
+    },
+    {
+      "epoch": 0.22402597402597402,
+      "grad_norm": 0.625,
+      "learning_rate": 1.9733559958825167e-05,
+      "loss": 0.3755,
+      "step": 69
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9704917941574053e-05,
+      "loss": 0.3733,
+      "step": 70
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "eval_loss": 0.3694484233856201,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9363,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 70
+    },
+    {
+      "epoch": 0.2305194805194805,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.9674836970574253e-05,
+      "loss": 0.3676,
+      "step": 71
+    },
+    {
+      "epoch": 0.23376623376623376,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9643321505948588e-05,
+      "loss": 0.3721,
+      "step": 72
+    },
+    {
+      "epoch": 0.237012987012987,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9610376220513067e-05,
+      "loss": 0.3558,
+      "step": 73
+    },
+    {
+      "epoch": 0.24025974025974026,
+      "grad_norm": 0.625,
+      "learning_rate": 1.957600599908406e-05,
+      "loss": 0.357,
+      "step": 74
+    },
+    {
+      "epoch": 0.2435064935064935,
+      "grad_norm": 0.66015625,
+      "learning_rate": 1.954021593775401e-05,
+      "loss": 0.3608,
+      "step": 75
+    },
+    {
+      "epoch": 0.24675324675324675,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9503011343135828e-05,
+      "loss": 0.3513,
+      "step": 76
+    },
+    {
+      "epoch": 0.25,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9464397731576093e-05,
+      "loss": 0.3693,
+      "step": 77
+    },
+    {
+      "epoch": 0.2532467532467532,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9424380828337146e-05,
+      "loss": 0.3605,
+      "step": 78
+    },
+    {
+      "epoch": 0.2564935064935065,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.938296656674817e-05,
+      "loss": 0.3553,
+      "step": 79
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.9340161087325483e-05,
+      "loss": 0.3721,
+      "step": 80
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "eval_loss": 0.3553767800331116,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0277,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 80
+    },
+    {
+      "epoch": 0.262987012987013,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9295970736862063e-05,
+      "loss": 0.3597,
+      "step": 81
+    },
+    {
+      "epoch": 0.2662337662337662,
+      "grad_norm": 0.65625,
+      "learning_rate": 1.9250402067486523e-05,
+      "loss": 0.3527,
+      "step": 82
+    },
+    {
+      "epoch": 0.2694805194805195,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9203461835691596e-05,
+      "loss": 0.3528,
+      "step": 83
+    },
+    {
+      "epoch": 0.2727272727272727,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.9155157001332374e-05,
+      "loss": 0.356,
+      "step": 84
+    },
+    {
+      "epoch": 0.275974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9105494726594344e-05,
+      "loss": 0.3556,
+      "step": 85
+    },
+    {
+      "epoch": 0.2792207792207792,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.905448237493147e-05,
+      "loss": 0.3473,
+      "step": 86
+    },
+    {
+      "epoch": 0.2824675324675325,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9002127509974376e-05,
+      "loss": 0.3441,
+      "step": 87
+    },
+    {
+      "epoch": 0.2857142857142857,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.894843789440892e-05,
+      "loss": 0.3492,
+      "step": 88
+    },
+    {
+      "epoch": 0.288961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.889342148882519e-05,
+      "loss": 0.3338,
+      "step": 89
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "grad_norm": 0.67578125,
+      "learning_rate": 1.8837086450537195e-05,
+      "loss": 0.3347,
+      "step": 90
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "eval_loss": 0.34332525730133057,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9257,
+      "eval_samples_per_second": 11.885,
+      "eval_steps_per_second": 1.486,
+      "step": 90
+    },
+    {
+      "epoch": 0.29545454545454547,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.877944113237336e-05,
+      "loss": 0.3438,
+      "step": 91
+    },
+    {
+      "epoch": 0.2987012987012987,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.872049408143808e-05,
+      "loss": 0.3505,
+      "step": 92
+    },
+    {
+      "epoch": 0.30194805194805197,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.866025403784439e-05,
+      "loss": 0.3362,
+      "step": 93
+    },
+    {
+      "epoch": 0.3051948051948052,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.8598729933418102e-05,
+      "loss": 0.337,
+      "step": 94
+    },
+    {
+      "epoch": 0.30844155844155846,
+      "grad_norm": 0.64453125,
+      "learning_rate": 1.8535930890373467e-05,
+      "loss": 0.3324,
+      "step": 95
+    },
+    {
+      "epoch": 0.3116883116883117,
+      "grad_norm": 0.6875,
+      "learning_rate": 1.8471866219960604e-05,
+      "loss": 0.3461,
+      "step": 96
+    },
+    {
+      "epoch": 0.31493506493506496,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.840654542108494e-05,
+      "loss": 0.3318,
+      "step": 97
+    },
+    {
+      "epoch": 0.3181818181818182,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.833997817889878e-05,
+      "loss": 0.3335,
+      "step": 98
+    },
+    {
+      "epoch": 0.32142857142857145,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.82721743633653e-05,
+      "loss": 0.3291,
+      "step": 99
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.820314402779511e-05,
+      "loss": 0.3324,
+      "step": 100
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "eval_loss": 0.33332687616348267,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9506,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 100
+    },
+    {
+      "epoch": 0.32792207792207795,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.8132897407355657e-05,
+      "loss": 0.3288,
+      "step": 101
+    },
+    {
+      "epoch": 0.33116883116883117,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.806144491755363e-05,
+      "loss": 0.3384,
+      "step": 102
+    },
+    {
+      "epoch": 0.3344155844155844,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.798879715269067e-05,
+      "loss": 0.3329,
+      "step": 103
+    },
+    {
+      "epoch": 0.33766233766233766,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.7914964884292543e-05,
+      "loss": 0.3329,
+      "step": 104
+    },
+    {
+      "epoch": 0.3409090909090909,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.7839959059512016e-05,
+      "loss": 0.3335,
+      "step": 105
+    },
+    {
+      "epoch": 0.34415584415584416,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.7763790799505746e-05,
+      "loss": 0.3201,
+      "step": 106
+    },
+    {
+      "epoch": 0.3474025974025974,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.7686471397785322e-05,
+      "loss": 0.3327,
+      "step": 107
+    },
+    {
+      "epoch": 0.35064935064935066,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.760801231854278e-05,
+      "loss": 0.3205,
+      "step": 108
+    },
+    {
+      "epoch": 0.3538961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.7528425194950794e-05,
+      "loss": 0.334,
+      "step": 109
+    },
+    {
+      "epoch": 0.35714285714285715,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.744772182743782e-05,
+      "loss": 0.3165,
+      "step": 110
+    },
+    {
+      "epoch": 0.35714285714285715,
+      "eval_loss": 0.32505103945732117,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0106,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 110
+    },
+    {
+      "epoch": 0.36038961038961037,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.736591418193844e-05,
+      "loss": 0.3236,
+      "step": 111
+    },
+    {
+      "epoch": 0.36363636363636365,
+      "grad_norm": 0.63671875,
+      "learning_rate": 1.728301438811916e-05,
+      "loss": 0.3192,
+      "step": 112
+    },
+    {
+      "epoch": 0.36688311688311687,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.7199034737579962e-05,
+      "loss": 0.327,
+      "step": 113
+    },
+    {
+      "epoch": 0.37012987012987014,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.711398768203178e-05,
+      "loss": 0.3169,
+      "step": 114
+    },
+    {
+      "epoch": 0.37337662337662336,
+      "grad_norm": 0.625,
+      "learning_rate": 1.7027885831450318e-05,
+      "loss": 0.3186,
+      "step": 115
+    },
+    {
+      "epoch": 0.37662337662337664,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.6940741952206342e-05,
+      "loss": 0.3136,
+      "step": 116
+    },
+    {
+      "epoch": 0.37987012987012986,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.6852568965172794e-05,
+      "loss": 0.3184,
+      "step": 117
+    },
+    {
+      "epoch": 0.38311688311688313,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.676337994380903e-05,
+      "loss": 0.3114,
+      "step": 118
+    },
+    {
+      "epoch": 0.38636363636363635,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.6673188112222394e-05,
+      "loss": 0.3329,
+      "step": 119
+    },
+    {
+      "epoch": 0.38961038961038963,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.658200684320748e-05,
+      "loss": 0.3105,
+      "step": 120
+    },
+    {
+      "epoch": 0.38961038961038963,
+      "eval_loss": 0.31796982884407043,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9896,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 120
+    },
+    {
+      "epoch": 0.39285714285714285,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.6489849656263336e-05,
+      "loss": 0.3084,
+      "step": 121
+    },
+    {
+      "epoch": 0.3961038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.6396730215588913e-05,
+      "loss": 0.3124,
+      "step": 122
+    },
+    {
+      "epoch": 0.39935064935064934,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.630266232805709e-05,
+      "loss": 0.3167,
+      "step": 123
+    },
+    {
+      "epoch": 0.4025974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.6207659941167485e-05,
+      "loss": 0.3084,
+      "step": 124
+    },
+    {
+      "epoch": 0.40584415584415584,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.6111737140978495e-05,
+      "loss": 0.3143,
+      "step": 125
+    },
+    {
+      "epoch": 0.4090909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.6014908150018703e-05,
+      "loss": 0.3173,
+      "step": 126
+    },
+    {
+      "epoch": 0.41233766233766234,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.591718732517814e-05,
+      "loss": 0.3199,
+      "step": 127
+    },
+    {
+      "epoch": 0.4155844155844156,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.581858915557953e-05,
+      "loss": 0.312,
+      "step": 128
+    },
+    {
+      "epoch": 0.41883116883116883,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.571912826043003e-05,
+      "loss": 0.3217,
+      "step": 129
+    },
+    {
+      "epoch": 0.42207792207792205,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.5618819386853607e-05,
+      "loss": 0.3123,
+      "step": 130
+    },
+    {
+      "epoch": 0.42207792207792205,
+      "eval_loss": 0.3119741976261139,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9669,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 130
+    },
+    {
+      "epoch": 0.4253246753246753,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.551767740770446e-05,
+      "loss": 0.3088,
+      "step": 131
+    },
+    {
+      "epoch": 0.42857142857142855,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.541571731936185e-05,
+      "loss": 0.3084,
+      "step": 132
+    },
+    {
+      "epoch": 0.4318181818181818,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5312954239506536e-05,
+      "loss": 0.3179,
+      "step": 133
+    },
+    {
+      "epoch": 0.43506493506493504,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.5209403404879305e-05,
+      "loss": 0.3145,
+      "step": 134
+    },
+    {
+      "epoch": 0.4383116883116883,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.5105080169021792e-05,
+      "loss": 0.3045,
+      "step": 135
+    },
+    {
+      "epoch": 0.44155844155844154,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.5000000000000002e-05,
+      "loss": 0.314,
+      "step": 136
+    },
+    {
+      "epoch": 0.4448051948051948,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.4894178478110856e-05,
+      "loss": 0.2985,
+      "step": 137
+    },
+    {
+      "epoch": 0.44805194805194803,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.4787631293572094e-05,
+      "loss": 0.3114,
+      "step": 138
+    },
+    {
+      "epoch": 0.4512987012987013,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.4680374244195861e-05,
+      "loss": 0.3129,
+      "step": 139
+    },
+    {
+      "epoch": 0.45454545454545453,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.4572423233046386e-05,
+      "loss": 0.3007,
+      "step": 140
+    },
+    {
+      "epoch": 0.45454545454545453,
+      "eval_loss": 0.30710509419441223,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9622,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 140
+    },
+    {
+      "epoch": 0.4577922077922078,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.4463794266081994e-05,
+      "loss": 0.2981,
+      "step": 141
+    },
+    {
+      "epoch": 0.461038961038961,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.4354503449781914e-05,
+      "loss": 0.2974,
+      "step": 142
+    },
+    {
+      "epoch": 0.4642857142857143,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.4244566988758152e-05,
+      "loss": 0.3064,
+      "step": 143
+    },
+    {
+      "epoch": 0.4675324675324675,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.4134001183352833e-05,
+      "loss": 0.298,
+      "step": 144
+    },
+    {
+      "epoch": 0.4707792207792208,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.4022822427221325e-05,
+      "loss": 0.3037,
+      "step": 145
+    },
+    {
+      "epoch": 0.474025974025974,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.391104720490156e-05,
+      "loss": 0.3062,
+      "step": 146
+    },
+    {
+      "epoch": 0.4772727272727273,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.3798692089369855e-05,
+      "loss": 0.3019,
+      "step": 147
+    },
+    {
+      "epoch": 0.4805194805194805,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.368577373958362e-05,
+      "loss": 0.3008,
+      "step": 148
+    },
+    {
+      "epoch": 0.4837662337662338,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.3572308898011328e-05,
+      "loss": 0.3055,
+      "step": 149
+    },
+    {
+      "epoch": 0.487012987012987,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.3458314388150115e-05,
+      "loss": 0.2942,
+      "step": 150
+    },
+    {
+      "epoch": 0.487012987012987,
+      "eval_loss": 0.30315953493118286,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9566,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 150
+    },
+    {
+      "epoch": 0.4902597402597403,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.3343807112031329e-05,
+      "loss": 0.3097,
+      "step": 151
+    },
+    {
+      "epoch": 0.4935064935064935,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.3228804047714462e-05,
+      "loss": 0.3004,
+      "step": 152
+    },
+    {
+      "epoch": 0.4967532467532468,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.3113322246769817e-05,
+      "loss": 0.3046,
+      "step": 153
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.2997378831750242e-05,
+      "loss": 0.2966,
+      "step": 154
+    },
+    {
+      "epoch": 0.5032467532467533,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.2880990993652379e-05,
+      "loss": 0.2989,
+      "step": 155
+    },
+    {
+      "epoch": 0.5064935064935064,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.2764175989367717e-05,
+      "loss": 0.3015,
+      "step": 156
+    },
+    {
+      "epoch": 0.5097402597402597,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.2646951139123935e-05,
+      "loss": 0.2976,
+      "step": 157
+    },
+    {
+      "epoch": 0.512987012987013,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.2529333823916807e-05,
+      "loss": 0.2919,
+      "step": 158
+    },
+    {
+      "epoch": 0.5162337662337663,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.241134148293311e-05,
+      "loss": 0.3041,
+      "step": 159
+    },
+    {
+      "epoch": 0.5194805194805194,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.2292991610964902e-05,
+      "loss": 0.3053,
+      "step": 160
+    },
+    {
+      "epoch": 0.5194805194805194,
+      "eval_loss": 0.3000594675540924,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9643,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 160
+    },
+    {
+      "epoch": 0.5227272727272727,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.2174301755815572e-05,
+      "loss": 0.3043,
+      "step": 161
+    },
+    {
+      "epoch": 0.525974025974026,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.2055289515698008e-05,
+      "loss": 0.2988,
+      "step": 162
+    },
+    {
+      "epoch": 0.5292207792207793,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1935972536625302e-05,
+      "loss": 0.2871,
+      "step": 163
+    },
+    {
+      "epoch": 0.5324675324675324,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.1816368509794365e-05,
+      "loss": 0.2985,
+      "step": 164
+    },
+    {
+      "epoch": 0.5357142857142857,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.1696495168962848e-05,
+      "loss": 0.3066,
+      "step": 165
+    },
+    {
+      "epoch": 0.538961038961039,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.1576370287819737e-05,
+      "loss": 0.296,
+      "step": 166
+    },
+    {
+      "epoch": 0.5422077922077922,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.1456011677350052e-05,
+      "loss": 0.2997,
+      "step": 167
+    },
+    {
+      "epoch": 0.5454545454545454,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.133543718319398e-05,
+      "loss": 0.3011,
+      "step": 168
+    },
+    {
+      "epoch": 0.5487012987012987,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1214664683000927e-05,
+      "loss": 0.2898,
+      "step": 169
+    },
+    {
+      "epoch": 0.551948051948052,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.1093712083778748e-05,
+      "loss": 0.292,
+      "step": 170
+    },
+    {
+      "epoch": 0.551948051948052,
+      "eval_loss": 0.29717138409614563,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9869,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 170
+    },
+    {
+      "epoch": 0.5551948051948052,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.0972597319238692e-05,
+      "loss": 0.2995,
+      "step": 171
+    },
+    {
+      "epoch": 0.5584415584415584,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.0851338347136358e-05,
+      "loss": 0.3015,
+      "step": 172
+    },
+    {
+      "epoch": 0.5616883116883117,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.0729953146609076e-05,
+      "loss": 0.2953,
+      "step": 173
+    },
+    {
+      "epoch": 0.564935064935065,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.060845971551014e-05,
+      "loss": 0.3011,
+      "step": 174
+    },
+    {
+      "epoch": 0.5681818181818182,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.0486876067740253e-05,
+      "loss": 0.3016,
+      "step": 175
+    },
+    {
+      "epoch": 0.5714285714285714,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.0365220230576592e-05,
+      "loss": 0.2953,
+      "step": 176
+    },
+    {
+      "epoch": 0.5746753246753247,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.0243510241999898e-05,
+      "loss": 0.2948,
+      "step": 177
+    },
+    {
+      "epoch": 0.577922077922078,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.0121764148019977e-05,
+      "loss": 0.2958,
+      "step": 178
+    },
+    {
+      "epoch": 0.5811688311688312,
+      "grad_norm": 0.5546875,
+      "learning_rate": 1e-05,
+      "loss": 0.2829,
+      "step": 179
+    },
+    {
+      "epoch": 0.5844155844155844,
+      "grad_norm": 0.58203125,
+      "learning_rate": 9.878235851980027e-06,
+      "loss": 0.2917,
+      "step": 180
+    },
+    {
+      "epoch": 0.5844155844155844,
+      "eval_loss": 0.29469791054725647,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9839,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 180
+    },
+    {
+      "epoch": 0.5876623376623377,
+      "grad_norm": 0.59375,
+      "learning_rate": 9.756489758000105e-06,
+      "loss": 0.2994,
+      "step": 181
+    },
+    {
+      "epoch": 0.5909090909090909,
+      "grad_norm": 0.57421875,
+      "learning_rate": 9.634779769423412e-06,
+      "loss": 0.3054,
+      "step": 182
+    },
+    {
+      "epoch": 0.5941558441558441,
+      "grad_norm": 0.59375,
+      "learning_rate": 9.51312393225975e-06,
+      "loss": 0.2925,
+      "step": 183
+    },
+    {
+      "epoch": 0.5974025974025974,
+      "grad_norm": 0.5703125,
+      "learning_rate": 9.391540284489862e-06,
+      "loss": 0.2816,
+      "step": 184
+    },
+    {
+      "epoch": 0.6006493506493507,
+      "grad_norm": 0.5390625,
+      "learning_rate": 9.270046853390924e-06,
+      "loss": 0.2856,
+      "step": 185
+    },
+    {
+      "epoch": 0.6038961038961039,
+      "grad_norm": 0.5703125,
+      "learning_rate": 9.148661652863644e-06,
+      "loss": 0.2936,
+      "step": 186
+    },
+    {
+      "epoch": 0.6071428571428571,
+      "grad_norm": 0.55078125,
+      "learning_rate": 9.02740268076131e-06,
+      "loss": 0.2938,
+      "step": 187
+    },
+    {
+      "epoch": 0.6103896103896104,
+      "grad_norm": 0.5859375,
+      "learning_rate": 8.906287916221259e-06,
+      "loss": 0.3017,
+      "step": 188
+    },
+    {
+      "epoch": 0.6136363636363636,
+      "grad_norm": 0.578125,
+      "learning_rate": 8.785335316999078e-06,
+      "loss": 0.2957,
+      "step": 189
+    },
+    {
+      "epoch": 0.6168831168831169,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.664562816806022e-06,
+      "loss": 0.2961,
+      "step": 190
+    },
+    {
+      "epoch": 0.6168831168831169,
+      "eval_loss": 0.2927069365978241,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9405,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 190
+    },
+    {
+      "epoch": 0.6201298701298701,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.543988322649954e-06,
+      "loss": 0.2915,
+      "step": 191
+    },
+    {
+      "epoch": 0.6233766233766234,
+      "grad_norm": 0.57421875,
+      "learning_rate": 8.423629712180265e-06,
+      "loss": 0.2875,
+      "step": 192
+    },
+    {
+      "epoch": 0.6266233766233766,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.303504831037154e-06,
+      "loss": 0.2979,
+      "step": 193
+    },
+    {
+      "epoch": 0.6298701298701299,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.183631490205636e-06,
+      "loss": 0.2891,
+      "step": 194
+    },
+    {
+      "epoch": 0.6331168831168831,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.064027463374702e-06,
+      "loss": 0.2796,
+      "step": 195
+    },
+    {
+      "epoch": 0.6363636363636364,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.944710484301995e-06,
+      "loss": 0.285,
+      "step": 196
+    },
+    {
+      "epoch": 0.6396103896103896,
+      "grad_norm": 0.58203125,
+      "learning_rate": 7.825698244184432e-06,
+      "loss": 0.295,
+      "step": 197
+    },
+    {
+      "epoch": 0.6428571428571429,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.707008389035102e-06,
+      "loss": 0.2962,
+      "step": 198
+    },
+    {
+      "epoch": 0.6461038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.588658517066893e-06,
+      "loss": 0.2912,
+      "step": 199
+    },
+    {
+      "epoch": 0.6493506493506493,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.470666176083193e-06,
+      "loss": 0.2917,
+      "step": 200
+    },
+    {
+      "epoch": 0.6493506493506493,
+      "eval_loss": 0.2912693917751312,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9875,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 200
+    },
+    {
+      "epoch": 0.6525974025974026,
+      "grad_norm": 0.6015625,
+      "learning_rate": 7.3530488608760645e-06,
+      "loss": 0.291,
+      "step": 201
+    },
+    {
+      "epoch": 0.6558441558441559,
+      "grad_norm": 0.5625,
+      "learning_rate": 7.235824010632284e-06,
+      "loss": 0.2862,
+      "step": 202
+    },
+    {
+      "epoch": 0.6590909090909091,
+      "grad_norm": 0.5546875,
+      "learning_rate": 7.119009006347625e-06,
+      "loss": 0.2936,
+      "step": 203
+    },
+    {
+      "epoch": 0.6623376623376623,
+      "grad_norm": 0.5546875,
+      "learning_rate": 7.002621168249759e-06,
+      "loss": 0.2934,
+      "step": 204
+    },
+    {
+      "epoch": 0.6655844155844156,
+      "grad_norm": 0.578125,
+      "learning_rate": 6.886677753230184e-06,
+      "loss": 0.2783,
+      "step": 205
+    },
+    {
+      "epoch": 0.6688311688311688,
+      "grad_norm": 0.546875,
+      "learning_rate": 6.771195952285541e-06,
+      "loss": 0.2844,
+      "step": 206
+    },
+    {
+      "epoch": 0.672077922077922,
+      "grad_norm": 0.5546875,
+      "learning_rate": 6.656192887968675e-06,
+      "loss": 0.2835,
+      "step": 207
+    },
+    {
+      "epoch": 0.6753246753246753,
+      "grad_norm": 0.56640625,
+      "learning_rate": 6.5416856118498874e-06,
+      "loss": 0.2887,
+      "step": 208
+    },
+    {
+      "epoch": 0.6785714285714286,
+      "grad_norm": 0.5625,
+      "learning_rate": 6.427691101988673e-06,
+      "loss": 0.2925,
+      "step": 209
+    },
+    {
+      "epoch": 0.6818181818181818,
+      "grad_norm": 0.58203125,
+      "learning_rate": 6.314226260416383e-06,
+      "loss": 0.29,
+      "step": 210
+    },
+    {
+      "epoch": 0.6818181818181818,
+      "eval_loss": 0.29015374183654785,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.8645,
+      "eval_samples_per_second": 11.889,
+      "eval_steps_per_second": 1.487,
+      "step": 210
+    },
+    {
+      "epoch": 0.685064935064935,
+      "grad_norm": 0.5625,
+      "learning_rate": 6.2013079106301454e-06,
+      "loss": 0.2858,
+      "step": 211
+    },
+    {
+      "epoch": 0.6883116883116883,
+      "grad_norm": 0.5703125,
+      "learning_rate": 6.088952795098442e-06,
+      "loss": 0.2828,
+      "step": 212
+    },
+    {
+      "epoch": 0.6915584415584416,
+      "grad_norm": 0.578125,
+      "learning_rate": 5.977177572778679e-06,
+      "loss": 0.2947,
+      "step": 213
+    },
+    {
+      "epoch": 0.6948051948051948,
+      "grad_norm": 0.578125,
+      "learning_rate": 5.8659988166471715e-06,
+      "loss": 0.2795,
+      "step": 214
+    },
+    {
+      "epoch": 0.698051948051948,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.755433011241851e-06,
+      "loss": 0.2899,
+      "step": 215
+    },
+    {
+      "epoch": 0.7012987012987013,
+      "grad_norm": 0.578125,
+      "learning_rate": 5.645496550218089e-06,
+      "loss": 0.2881,
+      "step": 216
+    },
+    {
+      "epoch": 0.7045454545454546,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.5362057339180075e-06,
+      "loss": 0.2937,
+      "step": 217
+    },
+    {
+      "epoch": 0.7077922077922078,
+      "grad_norm": 0.5625,
+      "learning_rate": 5.427576766953615e-06,
+      "loss": 0.2828,
+      "step": 218
+    },
+    {
+      "epoch": 0.711038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 5.3196257558041386e-06,
+      "loss": 0.2863,
+      "step": 219
+    },
+    {
+      "epoch": 0.7142857142857143,
+      "grad_norm": 0.5859375,
+      "learning_rate": 5.212368706427913e-06,
+      "loss": 0.2876,
+      "step": 220
+    },
+    {
+      "epoch": 0.7142857142857143,
+      "eval_loss": 0.2891963720321655,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9738,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 220
+    },
+    {
+      "epoch": 0.7175324675324676,
+      "grad_norm": 0.5859375,
+      "learning_rate": 5.105821521889147e-06,
+      "loss": 0.2955,
+      "step": 221
+    },
+    {
+      "epoch": 0.7207792207792207,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.000000000000003e-06,
+      "loss": 0.2873,
+      "step": 222
+    },
+    {
+      "epoch": 0.724025974025974,
+      "grad_norm": 0.5625,
+      "learning_rate": 4.894919830978212e-06,
+      "loss": 0.2813,
+      "step": 223
+    },
+    {
+      "epoch": 0.7272727272727273,
+      "grad_norm": 0.59375,
+      "learning_rate": 4.790596595120699e-06,
+      "loss": 0.2912,
+      "step": 224
+    },
+    {
+      "epoch": 0.7305194805194806,
+      "grad_norm": 0.5859375,
+      "learning_rate": 4.687045760493468e-06,
+      "loss": 0.2921,
+      "step": 225
+    },
+    {
+      "epoch": 0.7337662337662337,
+      "grad_norm": 0.55078125,
+      "learning_rate": 4.584282680638155e-06,
+      "loss": 0.288,
+      "step": 226
+    },
+    {
+      "epoch": 0.737012987012987,
+      "grad_norm": 0.5546875,
+      "learning_rate": 4.482322592295541e-06,
+      "loss": 0.2914,
+      "step": 227
+    },
+    {
+      "epoch": 0.7402597402597403,
+      "grad_norm": 0.56640625,
+      "learning_rate": 4.381180613146396e-06,
+      "loss": 0.2901,
+      "step": 228
+    },
+    {
+      "epoch": 0.7435064935064936,
+      "grad_norm": 0.56640625,
+      "learning_rate": 4.280871739569972e-06,
+      "loss": 0.2941,
+      "step": 229
+    },
+    {
+      "epoch": 0.7467532467532467,
+      "grad_norm": 0.54296875,
+      "learning_rate": 4.181410844420473e-06,
+      "loss": 0.2836,
+      "step": 230
+    },
+    {
+      "epoch": 0.7467532467532467,
+      "eval_loss": 0.2885654866695404,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9437,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 230
+    },
+    {
+      "epoch": 0.75,
+      "grad_norm": 0.58203125,
+      "learning_rate": 4.082812674821865e-06,
+      "loss": 0.2882,
+      "step": 231
+    },
+    {
+      "epoch": 0.7532467532467533,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.9850918499812976e-06,
+      "loss": 0.293,
+      "step": 232
+    },
+    {
+      "epoch": 0.7564935064935064,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.888262859021508e-06,
+      "loss": 0.2818,
+      "step": 233
+    },
+    {
+      "epoch": 0.7597402597402597,
+      "grad_norm": 0.5703125,
+      "learning_rate": 3.7923400588325156e-06,
+      "loss": 0.2836,
+      "step": 234
+    },
+    {
+      "epoch": 0.762987012987013,
+      "grad_norm": 0.6015625,
+      "learning_rate": 3.6973376719429134e-06,
+      "loss": 0.2833,
+      "step": 235
+    },
+    {
+      "epoch": 0.7662337662337663,
+      "grad_norm": 0.5625,
+      "learning_rate": 3.6032697844110896e-06,
+      "loss": 0.2956,
+      "step": 236
+    },
+    {
+      "epoch": 0.7694805194805194,
+      "grad_norm": 0.578125,
+      "learning_rate": 3.5101503437366678e-06,
+      "loss": 0.288,
+      "step": 237
+    },
+    {
+      "epoch": 0.7727272727272727,
+      "grad_norm": 0.578125,
+      "learning_rate": 3.4179931567925216e-06,
+      "loss": 0.2898,
+      "step": 238
+    },
+    {
+      "epoch": 0.775974025974026,
+      "grad_norm": 0.55859375,
+      "learning_rate": 3.326811887777607e-06,
+      "loss": 0.2809,
+      "step": 239
+    },
+    {
+      "epoch": 0.7792207792207793,
+      "grad_norm": 0.546875,
+      "learning_rate": 3.236620056190972e-06,
+      "loss": 0.285,
+      "step": 240
+    },
+    {
+      "epoch": 0.7792207792207793,
+      "eval_loss": 0.2881048321723938,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9387,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 240
+    },
+    {
+      "epoch": 0.7824675324675324,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.1474310348272084e-06,
+      "loss": 0.2888,
+      "step": 241
+    },
+    {
+      "epoch": 0.7857142857142857,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.0592580477936606e-06,
+      "loss": 0.2875,
+      "step": 242
+    },
+    {
+      "epoch": 0.788961038961039,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.9721141685496825e-06,
+      "loss": 0.2828,
+      "step": 243
+    },
+    {
+      "epoch": 0.7922077922077922,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.8860123179682244e-06,
+      "loss": 0.2866,
+      "step": 244
+    },
+    {
+      "epoch": 0.7954545454545454,
+      "grad_norm": 0.5703125,
+      "learning_rate": 2.8009652624200436e-06,
+      "loss": 0.289,
+      "step": 245
+    },
+    {
+      "epoch": 0.7987012987012987,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.7169856118808414e-06,
+      "loss": 0.2832,
+      "step": 246
+    },
+    {
+      "epoch": 0.801948051948052,
+      "grad_norm": 0.56640625,
+      "learning_rate": 2.634085818061565e-06,
+      "loss": 0.2913,
+      "step": 247
+    },
+    {
+      "epoch": 0.8051948051948052,
+      "grad_norm": 0.58203125,
+      "learning_rate": 2.5522781725621814e-06,
+      "loss": 0.2917,
+      "step": 248
+    },
+    {
+      "epoch": 0.8084415584415584,
+      "grad_norm": 0.56640625,
+      "learning_rate": 2.471574805049206e-06,
+      "loss": 0.2937,
+      "step": 249
+    },
+    {
+      "epoch": 0.8116883116883117,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.3919876814572197e-06,
+      "loss": 0.2932,
+      "step": 250
+    },
+    {
+      "epoch": 0.8116883116883117,
+      "eval_loss": 0.28785035014152527,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.993,
+      "eval_samples_per_second": 11.88,
+      "eval_steps_per_second": 1.486,
+      "step": 250
+    },
+    {
+      "epoch": 0.814935064935065,
+      "grad_norm": 0.56640625,
+      "learning_rate": 2.3135286022146785e-06,
+      "loss": 0.2973,
+      "step": 251
+    },
+    {
+      "epoch": 0.8181818181818182,
+      "grad_norm": 0.5703125,
+      "learning_rate": 2.2362092004942583e-06,
+      "loss": 0.2862,
+      "step": 252
+    },
+    {
+      "epoch": 0.8214285714285714,
+      "grad_norm": 0.57421875,
+      "learning_rate": 2.1600409404879875e-06,
+      "loss": 0.279,
+      "step": 253
+    },
+    {
+      "epoch": 0.8246753246753247,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.08503511570746e-06,
+      "loss": 0.2844,
+      "step": 254
+    },
+    {
+      "epoch": 0.827922077922078,
+      "grad_norm": 0.5546875,
+      "learning_rate": 2.0112028473093294e-06,
+      "loss": 0.2792,
+      "step": 255
+    },
+    {
+      "epoch": 0.8311688311688312,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.9385550824463727e-06,
+      "loss": 0.2921,
+      "step": 256
+    },
+    {
+      "epoch": 0.8344155844155844,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.8671025926443464e-06,
+      "loss": 0.2891,
+      "step": 257
+    },
+    {
+      "epoch": 0.8376623376623377,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.7968559722048906e-06,
+      "loss": 0.2848,
+      "step": 258
+    },
+    {
+      "epoch": 0.8409090909090909,
+      "grad_norm": 0.55078125,
+      "learning_rate": 1.7278256366347034e-06,
+      "loss": 0.2921,
+      "step": 259
+    },
+    {
+      "epoch": 0.8441558441558441,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.660021821101222e-06,
+      "loss": 0.2907,
+      "step": 260
+    },
+    {
+      "epoch": 0.8441558441558441,
+      "eval_loss": 0.2877131402492523,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.008,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 260
+    },
+    {
+      "epoch": 0.8474025974025974,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.5934545789150625e-06,
+      "loss": 0.2979,
+      "step": 261
+    },
+    {
+      "epoch": 0.8506493506493507,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.528133780039397e-06,
+      "loss": 0.2847,
+      "step": 262
+    },
+    {
+      "epoch": 0.8538961038961039,
+      "grad_norm": 0.55078125,
+      "learning_rate": 1.4640691096265358e-06,
+      "loss": 0.2902,
+      "step": 263
+    },
+    {
+      "epoch": 0.8571428571428571,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.401270066581899e-06,
+      "loss": 0.2861,
+      "step": 264
+    },
+    {
+      "epoch": 0.8603896103896104,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.339745962155613e-06,
+      "loss": 0.2823,
+      "step": 265
+    },
+    {
+      "epoch": 0.8636363636363636,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.279505918561923e-06,
+      "loss": 0.2855,
+      "step": 266
+    },
+    {
+      "epoch": 0.8668831168831169,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.220558867626639e-06,
+      "loss": 0.2853,
+      "step": 267
+    },
+    {
+      "epoch": 0.8701298701298701,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1629135494628097e-06,
+      "loss": 0.2891,
+      "step": 268
+    },
+    {
+      "epoch": 0.8733766233766234,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.1065785111748117e-06,
+      "loss": 0.2853,
+      "step": 269
+    },
+    {
+      "epoch": 0.8766233766233766,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.051562105591082e-06,
+      "loss": 0.292,
+      "step": 270
+    },
+    {
+      "epoch": 0.8766233766233766,
+      "eval_loss": 0.28765836358070374,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.932,
+      "eval_samples_per_second": 11.885,
+      "eval_steps_per_second": 1.486,
+      "step": 270
+    },
+    {
+      "epoch": 0.8798701298701299,
+      "grad_norm": 0.55859375,
+      "learning_rate": 9.978724900256265e-07,
+      "loss": 0.2825,
+      "step": 271
+    },
+    {
+      "epoch": 0.8831168831168831,
+      "grad_norm": 0.5546875,
+      "learning_rate": 9.455176250685338e-07,
+      "loss": 0.2836,
+      "step": 272
+    },
+    {
+      "epoch": 0.8863636363636364,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.945052734056581e-07,
+      "loss": 0.2959,
+      "step": 273
+    },
+    {
+      "epoch": 0.8896103896103896,
+      "grad_norm": 0.57421875,
+      "learning_rate": 8.448429986676298e-07,
+      "loss": 0.3003,
+      "step": 274
+    },
+    {
+      "epoch": 0.8928571428571429,
+      "grad_norm": 0.56640625,
+      "learning_rate": 7.965381643084069e-07,
+      "loss": 0.295,
+      "step": 275
+    },
+    {
+      "epoch": 0.8961038961038961,
+      "grad_norm": 0.5625,
+      "learning_rate": 7.495979325134806e-07,
+      "loss": 0.2873,
+      "step": 276
+    },
+    {
+      "epoch": 0.8993506493506493,
+      "grad_norm": 0.55078125,
+      "learning_rate": 7.040292631379386e-07,
+      "loss": 0.2827,
+      "step": 277
+    },
+    {
+      "epoch": 0.9025974025974026,
+      "grad_norm": 0.54296875,
+      "learning_rate": 6.598389126745209e-07,
+      "loss": 0.2797,
+      "step": 278
+    },
+    {
+      "epoch": 0.9058441558441559,
+      "grad_norm": 0.53125,
+      "learning_rate": 6.170334332518325e-07,
+      "loss": 0.272,
+      "step": 279
+    },
+    {
+      "epoch": 0.9090909090909091,
+      "grad_norm": 0.58203125,
+      "learning_rate": 5.756191716628556e-07,
+      "loss": 0.292,
+      "step": 280
+    },
+    {
+      "epoch": 0.9090909090909091,
+      "eval_loss": 0.28761279582977295,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9741,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 280
+    },
+    {
+      "epoch": 0.9123376623376623,
+      "grad_norm": 0.5703125,
+      "learning_rate": 5.35602268423906e-07,
+      "loss": 0.2961,
+      "step": 281
+    },
+    {
+      "epoch": 0.9155844155844156,
+      "grad_norm": 0.578125,
+      "learning_rate": 4.969886568641757e-07,
+      "loss": 0.2904,
+      "step": 282
+    },
+    {
+      "epoch": 0.9188311688311688,
+      "grad_norm": 0.5625,
+      "learning_rate": 4.597840622459937e-07,
+      "loss": 0.2874,
+      "step": 283
+    },
+    {
+      "epoch": 0.922077922077922,
+      "grad_norm": 0.546875,
+      "learning_rate": 4.2399400091594154e-07,
+      "loss": 0.2794,
+      "step": 284
+    },
+    {
+      "epoch": 0.9253246753246753,
+      "grad_norm": 0.578125,
+      "learning_rate": 3.8962377948693395e-07,
+      "loss": 0.2962,
+      "step": 285
+    },
+    {
+      "epoch": 0.9285714285714286,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.566784940514145e-07,
+      "loss": 0.2914,
+      "step": 286
+    },
+    {
+      "epoch": 0.9318181818181818,
+      "grad_norm": 0.5703125,
+      "learning_rate": 3.2516302942574794e-07,
+      "loss": 0.2838,
+      "step": 287
+    },
+    {
+      "epoch": 0.935064935064935,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.9508205842594727e-07,
+      "loss": 0.2812,
+      "step": 288
+    },
+    {
+      "epoch": 0.9383116883116883,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.6644004117483357e-07,
+      "loss": 0.2827,
+      "step": 289
+    },
+    {
+      "epoch": 0.9415584415584416,
+      "grad_norm": 0.5546875,
+      "learning_rate": 2.392412244407294e-07,
+      "loss": 0.2791,
+      "step": 290
+    },
+    {
+      "epoch": 0.9415584415584416,
+      "eval_loss": 0.28759849071502686,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0075,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 290
+    },
+    {
+      "epoch": 0.9448051948051948,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.1348964100778914e-07,
+      "loss": 0.285,
+      "step": 291
+    },
+    {
+      "epoch": 0.948051948051948,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.8918910907805733e-07,
+      "loss": 0.288,
+      "step": 292
+    },
+    {
+      "epoch": 0.9512987012987013,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.6634323170533928e-07,
+      "loss": 0.2854,
+      "step": 293
+    },
+    {
+      "epoch": 0.9545454545454546,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.4495539626097289e-07,
+      "loss": 0.2873,
+      "step": 294
+    },
+    {
+      "epoch": 0.9577922077922078,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.2502877393158587e-07,
+      "loss": 0.2841,
+      "step": 295
+    },
+    {
+      "epoch": 0.961038961038961,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.0656631924889749e-07,
+      "loss": 0.2994,
+      "step": 296
+    },
+    {
+      "epoch": 0.9642857142857143,
+      "grad_norm": 0.5546875,
+      "learning_rate": 8.957076965165234e-08,
+      "loss": 0.2797,
+      "step": 297
+    },
+    {
+      "epoch": 0.9675324675324676,
+      "grad_norm": 0.56640625,
+      "learning_rate": 7.404464507973608e-08,
+      "loss": 0.2868,
+      "step": 298
+    },
+    {
+      "epoch": 0.9707792207792207,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.999024760054095e-08,
+      "loss": 0.2867,
+      "step": 299
+    },
+    {
+      "epoch": 0.974025974025974,
+      "grad_norm": 0.56640625,
+      "learning_rate": 4.740966106764222e-08,
+      "loss": 0.2832,
+      "step": 300
+    },
+    {
+      "epoch": 0.974025974025974,
+      "eval_loss": 0.287624329328537,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9045,
+      "eval_samples_per_second": 11.886,
+      "eval_steps_per_second": 1.487,
+      "step": 300
+    }
+  ],
+  "logging_steps": 1,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 100,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1.7706338758950912e+18,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/checkpoint-300/training_args.bin b/checkpoint-300/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..4684d3c90558d6bfde405c82b58596bc6231bfee
--- /dev/null
+++ b/checkpoint-300/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d09f69be12f7ef607f0a1508435a517f5277e3160e5d6a59346e619fcbaa3592
+size 5176
diff --git a/checkpoint-308/config.json b/checkpoint-308/config.json
new file mode 100644
index 0000000000000000000000000000000000000000..de55730a672c2805e1108109d81a111b0cdb48b1
--- /dev/null
+++ b/checkpoint-308/config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f8b621112eddc9da772fae3f70d672f4d949cc5cb1d580f135d66b3836598c1a
+size 915
diff --git a/checkpoint-308/generation_config.json b/checkpoint-308/generation_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..1007819213eaa3e36dd0c6354016734419f4a863
--- /dev/null
+++ b/checkpoint-308/generation_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a7327788370b7783e29852cf19b4a1f40cd31af5962c94bba76c336d491b4169
+size 184
diff --git a/checkpoint-308/model-00001-of-00004.safetensors b/checkpoint-308/model-00001-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..616bb67b1f5310e3c9955e638763bdeff0fb55c8
--- /dev/null
+++ b/checkpoint-308/model-00001-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3abfe37c12d3907a891b3b7282c09590bedc9a560e05339e0aebfba26341dc1b
+size 4976698672
diff --git a/checkpoint-308/model-00002-of-00004.safetensors b/checkpoint-308/model-00002-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..4c1a59736868aa002e16da3948c4b68be6d97a71
--- /dev/null
+++ b/checkpoint-308/model-00002-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:584021d05e0ffd6879fcb1af88d896f418db0a6656a6c799d17e6a28e6540c8b
+size 4999802720
diff --git a/checkpoint-308/model-00003-of-00004.safetensors b/checkpoint-308/model-00003-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..075dada5dcfe243936f5583203b90ba80f0935e1
--- /dev/null
+++ b/checkpoint-308/model-00003-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ced290d33c8beefb926d706c895d70bd58e0d89ba90daca1d637367a6e9163ae
+size 4915916176
diff --git a/checkpoint-308/model-00004-of-00004.safetensors b/checkpoint-308/model-00004-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..b1e38fdf23d6ff664ce00189e3dc1ab419ed8545
--- /dev/null
+++ b/checkpoint-308/model-00004-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:03dad76ab9974b3c84fcd8067e459ff22c49e10e1408ea16637a06df0b9b71c4
+size 1168138808
diff --git a/checkpoint-308/model.safetensors.index.json b/checkpoint-308/model.safetensors.index.json
new file mode 100644
index 0000000000000000000000000000000000000000..a054aad8cf0cb5671e2e10b3bf817725a4bec031
--- /dev/null
+++ b/checkpoint-308/model.safetensors.index.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:146776fce3f6db1103aa6f249e65ee5544c5923ce6f971b092eee79aa6e5d37b
+size 23950
diff --git a/checkpoint-308/optimizer.pt b/checkpoint-308/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..81d9520226370e1855c92fe28bb828c9ac7e5cdc
--- /dev/null
+++ b/checkpoint-308/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2b7c39b809a656f112d5f086526a410b408bd0992a7598ccf39a1595172d1717
+size 32121299754
diff --git a/checkpoint-308/rng_state.pth b/checkpoint-308/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..02e6a75cccce6d8b138f83bf61f04a2d82786ce4
--- /dev/null
+++ b/checkpoint-308/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:374644995be79c967030295cf4e5929f6a33d3b1197b8855bc39718b44686552
+size 14244
diff --git a/checkpoint-308/scheduler.pt b/checkpoint-308/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..cfb0140e43a2ea276537b0c933e85d359f3ef68f
--- /dev/null
+++ b/checkpoint-308/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c7d6a70f369c23098e0581c238ab3095159ad5894f8b6e0d4e2094eec5204ed6
+size 1064
diff --git a/checkpoint-308/special_tokens_map.json b/checkpoint-308/special_tokens_map.json
new file mode 100644
index 0000000000000000000000000000000000000000..92d7fab6729d723f45689803b99006fec945cc47
--- /dev/null
+++ b/checkpoint-308/special_tokens_map.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b1835caa5b4d70acaa210fa222b0036f1882f9525c4660fd4810fb3e1e40ff8
+size 325
diff --git a/checkpoint-308/tokenizer.json b/checkpoint-308/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..66cd9d7e0daec95eb10d16a63c615637dbbb7304
--- /dev/null
+++ b/checkpoint-308/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79e3e522635f3171300913bb421464a87de6222182a0570b9b2ccba2a964b2b4
+size 9085657
diff --git a/checkpoint-308/tokenizer_config.json b/checkpoint-308/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..08ebc9598183a9a76c44ef06813aca75f51de4a5
--- /dev/null
+++ b/checkpoint-308/tokenizer_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:05d0de6d831b5940c3f2eae4e495b9f77612cbb05b5804846733ab617f0dfe9b
+size 55380
diff --git a/checkpoint-308/trainer_state.json b/checkpoint-308/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..959072709d27771c0774876b2d02d45de908928e
--- /dev/null
+++ b/checkpoint-308/trainer_state.json
@@ -0,0 +1,2459 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 10,
+  "global_step": 308,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.003246753246753247,
+      "grad_norm": 16.375,
+      "learning_rate": 4.0000000000000003e-07,
+      "loss": 1.4017,
+      "step": 1
+    },
+    {
+      "epoch": 0.006493506493506494,
+      "grad_norm": 16.25,
+      "learning_rate": 8.000000000000001e-07,
+      "loss": 1.4229,
+      "step": 2
+    },
+    {
+      "epoch": 0.00974025974025974,
+      "grad_norm": 16.75,
+      "learning_rate": 1.2000000000000002e-06,
+      "loss": 1.4112,
+      "step": 3
+    },
+    {
+      "epoch": 0.012987012987012988,
+      "grad_norm": 16.5,
+      "learning_rate": 1.6000000000000001e-06,
+      "loss": 1.3959,
+      "step": 4
+    },
+    {
+      "epoch": 0.016233766233766232,
+      "grad_norm": 16.125,
+      "learning_rate": 2.0000000000000003e-06,
+      "loss": 1.3789,
+      "step": 5
+    },
+    {
+      "epoch": 0.01948051948051948,
+      "grad_norm": 14.4375,
+      "learning_rate": 2.4000000000000003e-06,
+      "loss": 1.3618,
+      "step": 6
+    },
+    {
+      "epoch": 0.022727272727272728,
+      "grad_norm": 10.1875,
+      "learning_rate": 2.8000000000000003e-06,
+      "loss": 1.3214,
+      "step": 7
+    },
+    {
+      "epoch": 0.025974025974025976,
+      "grad_norm": 7.4375,
+      "learning_rate": 3.2000000000000003e-06,
+      "loss": 1.2885,
+      "step": 8
+    },
+    {
+      "epoch": 0.02922077922077922,
+      "grad_norm": 6.28125,
+      "learning_rate": 3.6000000000000003e-06,
+      "loss": 1.2436,
+      "step": 9
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "grad_norm": 5.75,
+      "learning_rate": 4.000000000000001e-06,
+      "loss": 1.2269,
+      "step": 10
+    },
+    {
+      "epoch": 0.032467532467532464,
+      "eval_loss": 1.1976113319396973,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0273,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 10
+    },
+    {
+      "epoch": 0.03571428571428571,
+      "grad_norm": 5.375,
+      "learning_rate": 4.4e-06,
+      "loss": 1.1925,
+      "step": 11
+    },
+    {
+      "epoch": 0.03896103896103896,
+      "grad_norm": 5.125,
+      "learning_rate": 4.800000000000001e-06,
+      "loss": 1.1307,
+      "step": 12
+    },
+    {
+      "epoch": 0.04220779220779221,
+      "grad_norm": 5.125,
+      "learning_rate": 5.2e-06,
+      "loss": 1.0778,
+      "step": 13
+    },
+    {
+      "epoch": 0.045454545454545456,
+      "grad_norm": 5.0625,
+      "learning_rate": 5.600000000000001e-06,
+      "loss": 1.0067,
+      "step": 14
+    },
+    {
+      "epoch": 0.048701298701298704,
+      "grad_norm": 5.75,
+      "learning_rate": 6e-06,
+      "loss": 0.9668,
+      "step": 15
+    },
+    {
+      "epoch": 0.05194805194805195,
+      "grad_norm": 5.15625,
+      "learning_rate": 6.4000000000000006e-06,
+      "loss": 0.875,
+      "step": 16
+    },
+    {
+      "epoch": 0.05519480519480519,
+      "grad_norm": 4.875,
+      "learning_rate": 6.800000000000001e-06,
+      "loss": 0.8297,
+      "step": 17
+    },
+    {
+      "epoch": 0.05844155844155844,
+      "grad_norm": 4.5625,
+      "learning_rate": 7.2000000000000005e-06,
+      "loss": 0.7703,
+      "step": 18
+    },
+    {
+      "epoch": 0.06168831168831169,
+      "grad_norm": 3.90625,
+      "learning_rate": 7.600000000000001e-06,
+      "loss": 0.7327,
+      "step": 19
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "grad_norm": 2.78125,
+      "learning_rate": 8.000000000000001e-06,
+      "loss": 0.6672,
+      "step": 20
+    },
+    {
+      "epoch": 0.06493506493506493,
+      "eval_loss": 0.6521233320236206,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.999,
+      "eval_samples_per_second": 11.88,
+      "eval_steps_per_second": 1.486,
+      "step": 20
+    },
+    {
+      "epoch": 0.06818181818181818,
+      "grad_norm": 2.140625,
+      "learning_rate": 8.400000000000001e-06,
+      "loss": 0.6507,
+      "step": 21
+    },
+    {
+      "epoch": 0.07142857142857142,
+      "grad_norm": 1.828125,
+      "learning_rate": 8.8e-06,
+      "loss": 0.6215,
+      "step": 22
+    },
+    {
+      "epoch": 0.07467532467532467,
+      "grad_norm": 1.5546875,
+      "learning_rate": 9.200000000000002e-06,
+      "loss": 0.5992,
+      "step": 23
+    },
+    {
+      "epoch": 0.07792207792207792,
+      "grad_norm": 1.4453125,
+      "learning_rate": 9.600000000000001e-06,
+      "loss": 0.5748,
+      "step": 24
+    },
+    {
+      "epoch": 0.08116883116883117,
+      "grad_norm": 1.328125,
+      "learning_rate": 1e-05,
+      "loss": 0.5572,
+      "step": 25
+    },
+    {
+      "epoch": 0.08441558441558442,
+      "grad_norm": 1.4765625,
+      "learning_rate": 1.04e-05,
+      "loss": 0.542,
+      "step": 26
+    },
+    {
+      "epoch": 0.08766233766233766,
+      "grad_norm": 1.1171875,
+      "learning_rate": 1.0800000000000002e-05,
+      "loss": 0.5317,
+      "step": 27
+    },
+    {
+      "epoch": 0.09090909090909091,
+      "grad_norm": 0.90625,
+      "learning_rate": 1.1200000000000001e-05,
+      "loss": 0.5173,
+      "step": 28
+    },
+    {
+      "epoch": 0.09415584415584416,
+      "grad_norm": 0.83984375,
+      "learning_rate": 1.16e-05,
+      "loss": 0.4892,
+      "step": 29
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "grad_norm": 0.828125,
+      "learning_rate": 1.2e-05,
+      "loss": 0.491,
+      "step": 30
+    },
+    {
+      "epoch": 0.09740259740259741,
+      "eval_loss": 0.48583006858825684,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9705,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 30
+    },
+    {
+      "epoch": 0.10064935064935066,
+      "grad_norm": 0.8359375,
+      "learning_rate": 1.2400000000000002e-05,
+      "loss": 0.4852,
+      "step": 31
+    },
+    {
+      "epoch": 0.1038961038961039,
+      "grad_norm": 1.0078125,
+      "learning_rate": 1.2800000000000001e-05,
+      "loss": 0.4749,
+      "step": 32
+    },
+    {
+      "epoch": 0.10714285714285714,
+      "grad_norm": 0.99609375,
+      "learning_rate": 1.3200000000000002e-05,
+      "loss": 0.467,
+      "step": 33
+    },
+    {
+      "epoch": 0.11038961038961038,
+      "grad_norm": 0.734375,
+      "learning_rate": 1.3600000000000002e-05,
+      "loss": 0.4545,
+      "step": 34
+    },
+    {
+      "epoch": 0.11363636363636363,
+      "grad_norm": 0.69140625,
+      "learning_rate": 1.4e-05,
+      "loss": 0.4609,
+      "step": 35
+    },
+    {
+      "epoch": 0.11688311688311688,
+      "grad_norm": 0.69921875,
+      "learning_rate": 1.4400000000000001e-05,
+      "loss": 0.4497,
+      "step": 36
+    },
+    {
+      "epoch": 0.12012987012987013,
+      "grad_norm": 0.703125,
+      "learning_rate": 1.48e-05,
+      "loss": 0.4607,
+      "step": 37
+    },
+    {
+      "epoch": 0.12337662337662338,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5200000000000002e-05,
+      "loss": 0.443,
+      "step": 38
+    },
+    {
+      "epoch": 0.1266233766233766,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.5600000000000003e-05,
+      "loss": 0.4428,
+      "step": 39
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.6000000000000003e-05,
+      "loss": 0.435,
+      "step": 40
+    },
+    {
+      "epoch": 0.12987012987012986,
+      "eval_loss": 0.4367639422416687,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0177,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 40
+    },
+    {
+      "epoch": 0.1331168831168831,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.64e-05,
+      "loss": 0.438,
+      "step": 41
+    },
+    {
+      "epoch": 0.13636363636363635,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.6800000000000002e-05,
+      "loss": 0.4392,
+      "step": 42
+    },
+    {
+      "epoch": 0.1396103896103896,
+      "grad_norm": 0.65234375,
+      "learning_rate": 1.72e-05,
+      "loss": 0.435,
+      "step": 43
+    },
+    {
+      "epoch": 0.14285714285714285,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.76e-05,
+      "loss": 0.4356,
+      "step": 44
+    },
+    {
+      "epoch": 0.1461038961038961,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.8e-05,
+      "loss": 0.4172,
+      "step": 45
+    },
+    {
+      "epoch": 0.14935064935064934,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.8400000000000003e-05,
+      "loss": 0.4162,
+      "step": 46
+    },
+    {
+      "epoch": 0.1525974025974026,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.88e-05,
+      "loss": 0.418,
+      "step": 47
+    },
+    {
+      "epoch": 0.15584415584415584,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.9200000000000003e-05,
+      "loss": 0.4177,
+      "step": 48
+    },
+    {
+      "epoch": 0.1590909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9600000000000002e-05,
+      "loss": 0.4064,
+      "step": 49
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "grad_norm": 0.62890625,
+      "learning_rate": 2e-05,
+      "loss": 0.4078,
+      "step": 50
+    },
+    {
+      "epoch": 0.16233766233766234,
+      "eval_loss": 0.4077843129634857,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.8482,
+      "eval_samples_per_second": 11.89,
+      "eval_steps_per_second": 1.487,
+      "step": 50
+    },
+    {
+      "epoch": 0.16558441558441558,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9999258647132645e-05,
+      "loss": 0.4057,
+      "step": 51
+    },
+    {
+      "epoch": 0.16883116883116883,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.9997034698451396e-05,
+      "loss": 0.4172,
+      "step": 52
+    },
+    {
+      "epoch": 0.17207792207792208,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9993328483702393e-05,
+      "loss": 0.4082,
+      "step": 53
+    },
+    {
+      "epoch": 0.17532467532467533,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998814055240823e-05,
+      "loss": 0.3973,
+      "step": 54
+    },
+    {
+      "epoch": 0.17857142857142858,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.998147167378645e-05,
+      "loss": 0.3969,
+      "step": 55
+    },
+    {
+      "epoch": 0.18181818181818182,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9973322836635517e-05,
+      "loss": 0.4032,
+      "step": 56
+    },
+    {
+      "epoch": 0.18506493506493507,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9963695249188185e-05,
+      "loss": 0.3956,
+      "step": 57
+    },
+    {
+      "epoch": 0.18831168831168832,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.995259033893236e-05,
+      "loss": 0.3952,
+      "step": 58
+    },
+    {
+      "epoch": 0.19155844155844157,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9940009752399462e-05,
+      "loss": 0.3905,
+      "step": 59
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9925955354920265e-05,
+      "loss": 0.3822,
+      "step": 60
+    },
+    {
+      "epoch": 0.19480519480519481,
+      "eval_loss": 0.38575297594070435,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9122,
+      "eval_samples_per_second": 11.886,
+      "eval_steps_per_second": 1.486,
+      "step": 60
+    },
+    {
+      "epoch": 0.19805194805194806,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9910429230348348e-05,
+      "loss": 0.3976,
+      "step": 61
+    },
+    {
+      "epoch": 0.2012987012987013,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9893433680751105e-05,
+      "loss": 0.3714,
+      "step": 62
+    },
+    {
+      "epoch": 0.20454545454545456,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9874971226068417e-05,
+      "loss": 0.3739,
+      "step": 63
+    },
+    {
+      "epoch": 0.2077922077922078,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.985504460373903e-05,
+      "loss": 0.3907,
+      "step": 64
+    },
+    {
+      "epoch": 0.21103896103896103,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.983365676829466e-05,
+      "loss": 0.3803,
+      "step": 65
+    },
+    {
+      "epoch": 0.21428571428571427,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.9810810890921943e-05,
+      "loss": 0.3902,
+      "step": 66
+    },
+    {
+      "epoch": 0.21753246753246752,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.9786510358992213e-05,
+      "loss": 0.3744,
+      "step": 67
+    },
+    {
+      "epoch": 0.22077922077922077,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.9760758775559275e-05,
+      "loss": 0.3668,
+      "step": 68
+    },
+    {
+      "epoch": 0.22402597402597402,
+      "grad_norm": 0.625,
+      "learning_rate": 1.9733559958825167e-05,
+      "loss": 0.3755,
+      "step": 69
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9704917941574053e-05,
+      "loss": 0.3733,
+      "step": 70
+    },
+    {
+      "epoch": 0.22727272727272727,
+      "eval_loss": 0.3694484233856201,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9363,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 70
+    },
+    {
+      "epoch": 0.2305194805194805,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.9674836970574253e-05,
+      "loss": 0.3676,
+      "step": 71
+    },
+    {
+      "epoch": 0.23376623376623376,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9643321505948588e-05,
+      "loss": 0.3721,
+      "step": 72
+    },
+    {
+      "epoch": 0.237012987012987,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.9610376220513067e-05,
+      "loss": 0.3558,
+      "step": 73
+    },
+    {
+      "epoch": 0.24025974025974026,
+      "grad_norm": 0.625,
+      "learning_rate": 1.957600599908406e-05,
+      "loss": 0.357,
+      "step": 74
+    },
+    {
+      "epoch": 0.2435064935064935,
+      "grad_norm": 0.66015625,
+      "learning_rate": 1.954021593775401e-05,
+      "loss": 0.3608,
+      "step": 75
+    },
+    {
+      "epoch": 0.24675324675324675,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9503011343135828e-05,
+      "loss": 0.3513,
+      "step": 76
+    },
+    {
+      "epoch": 0.25,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9464397731576093e-05,
+      "loss": 0.3693,
+      "step": 77
+    },
+    {
+      "epoch": 0.2532467532467532,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.9424380828337146e-05,
+      "loss": 0.3605,
+      "step": 78
+    },
+    {
+      "epoch": 0.2564935064935065,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.938296656674817e-05,
+      "loss": 0.3553,
+      "step": 79
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.9340161087325483e-05,
+      "loss": 0.3721,
+      "step": 80
+    },
+    {
+      "epoch": 0.2597402597402597,
+      "eval_loss": 0.3553767800331116,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0277,
+      "eval_samples_per_second": 11.878,
+      "eval_steps_per_second": 1.485,
+      "step": 80
+    },
+    {
+      "epoch": 0.262987012987013,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.9295970736862063e-05,
+      "loss": 0.3597,
+      "step": 81
+    },
+    {
+      "epoch": 0.2662337662337662,
+      "grad_norm": 0.65625,
+      "learning_rate": 1.9250402067486523e-05,
+      "loss": 0.3527,
+      "step": 82
+    },
+    {
+      "epoch": 0.2694805194805195,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.9203461835691596e-05,
+      "loss": 0.3528,
+      "step": 83
+    },
+    {
+      "epoch": 0.2727272727272727,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.9155157001332374e-05,
+      "loss": 0.356,
+      "step": 84
+    },
+    {
+      "epoch": 0.275974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.9105494726594344e-05,
+      "loss": 0.3556,
+      "step": 85
+    },
+    {
+      "epoch": 0.2792207792207792,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.905448237493147e-05,
+      "loss": 0.3473,
+      "step": 86
+    },
+    {
+      "epoch": 0.2824675324675325,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.9002127509974376e-05,
+      "loss": 0.3441,
+      "step": 87
+    },
+    {
+      "epoch": 0.2857142857142857,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.894843789440892e-05,
+      "loss": 0.3492,
+      "step": 88
+    },
+    {
+      "epoch": 0.288961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.889342148882519e-05,
+      "loss": 0.3338,
+      "step": 89
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "grad_norm": 0.67578125,
+      "learning_rate": 1.8837086450537195e-05,
+      "loss": 0.3347,
+      "step": 90
+    },
+    {
+      "epoch": 0.2922077922077922,
+      "eval_loss": 0.34332525730133057,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9257,
+      "eval_samples_per_second": 11.885,
+      "eval_steps_per_second": 1.486,
+      "step": 90
+    },
+    {
+      "epoch": 0.29545454545454547,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.877944113237336e-05,
+      "loss": 0.3438,
+      "step": 91
+    },
+    {
+      "epoch": 0.2987012987012987,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.872049408143808e-05,
+      "loss": 0.3505,
+      "step": 92
+    },
+    {
+      "epoch": 0.30194805194805197,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.866025403784439e-05,
+      "loss": 0.3362,
+      "step": 93
+    },
+    {
+      "epoch": 0.3051948051948052,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.8598729933418102e-05,
+      "loss": 0.337,
+      "step": 94
+    },
+    {
+      "epoch": 0.30844155844155846,
+      "grad_norm": 0.64453125,
+      "learning_rate": 1.8535930890373467e-05,
+      "loss": 0.3324,
+      "step": 95
+    },
+    {
+      "epoch": 0.3116883116883117,
+      "grad_norm": 0.6875,
+      "learning_rate": 1.8471866219960604e-05,
+      "loss": 0.3461,
+      "step": 96
+    },
+    {
+      "epoch": 0.31493506493506496,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.840654542108494e-05,
+      "loss": 0.3318,
+      "step": 97
+    },
+    {
+      "epoch": 0.3181818181818182,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.833997817889878e-05,
+      "loss": 0.3335,
+      "step": 98
+    },
+    {
+      "epoch": 0.32142857142857145,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.82721743633653e-05,
+      "loss": 0.3291,
+      "step": 99
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.820314402779511e-05,
+      "loss": 0.3324,
+      "step": 100
+    },
+    {
+      "epoch": 0.3246753246753247,
+      "eval_loss": 0.33332687616348267,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9506,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 100
+    },
+    {
+      "epoch": 0.32792207792207795,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.8132897407355657e-05,
+      "loss": 0.3288,
+      "step": 101
+    },
+    {
+      "epoch": 0.33116883116883117,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.806144491755363e-05,
+      "loss": 0.3384,
+      "step": 102
+    },
+    {
+      "epoch": 0.3344155844155844,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.798879715269067e-05,
+      "loss": 0.3329,
+      "step": 103
+    },
+    {
+      "epoch": 0.33766233766233766,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.7914964884292543e-05,
+      "loss": 0.3329,
+      "step": 104
+    },
+    {
+      "epoch": 0.3409090909090909,
+      "grad_norm": 0.671875,
+      "learning_rate": 1.7839959059512016e-05,
+      "loss": 0.3335,
+      "step": 105
+    },
+    {
+      "epoch": 0.34415584415584416,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.7763790799505746e-05,
+      "loss": 0.3201,
+      "step": 106
+    },
+    {
+      "epoch": 0.3474025974025974,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.7686471397785322e-05,
+      "loss": 0.3327,
+      "step": 107
+    },
+    {
+      "epoch": 0.35064935064935066,
+      "grad_norm": 0.640625,
+      "learning_rate": 1.760801231854278e-05,
+      "loss": 0.3205,
+      "step": 108
+    },
+    {
+      "epoch": 0.3538961038961039,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.7528425194950794e-05,
+      "loss": 0.334,
+      "step": 109
+    },
+    {
+      "epoch": 0.35714285714285715,
+      "grad_norm": 0.6015625,
+      "learning_rate": 1.744772182743782e-05,
+      "loss": 0.3165,
+      "step": 110
+    },
+    {
+      "epoch": 0.35714285714285715,
+      "eval_loss": 0.32505103945732117,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0106,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 110
+    },
+    {
+      "epoch": 0.36038961038961037,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.736591418193844e-05,
+      "loss": 0.3236,
+      "step": 111
+    },
+    {
+      "epoch": 0.36363636363636365,
+      "grad_norm": 0.63671875,
+      "learning_rate": 1.728301438811916e-05,
+      "loss": 0.3192,
+      "step": 112
+    },
+    {
+      "epoch": 0.36688311688311687,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.7199034737579962e-05,
+      "loss": 0.327,
+      "step": 113
+    },
+    {
+      "epoch": 0.37012987012987014,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.711398768203178e-05,
+      "loss": 0.3169,
+      "step": 114
+    },
+    {
+      "epoch": 0.37337662337662336,
+      "grad_norm": 0.625,
+      "learning_rate": 1.7027885831450318e-05,
+      "loss": 0.3186,
+      "step": 115
+    },
+    {
+      "epoch": 0.37662337662337664,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.6940741952206342e-05,
+      "loss": 0.3136,
+      "step": 116
+    },
+    {
+      "epoch": 0.37987012987012986,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.6852568965172794e-05,
+      "loss": 0.3184,
+      "step": 117
+    },
+    {
+      "epoch": 0.38311688311688313,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.676337994380903e-05,
+      "loss": 0.3114,
+      "step": 118
+    },
+    {
+      "epoch": 0.38636363636363635,
+      "grad_norm": 0.6640625,
+      "learning_rate": 1.6673188112222394e-05,
+      "loss": 0.3329,
+      "step": 119
+    },
+    {
+      "epoch": 0.38961038961038963,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.658200684320748e-05,
+      "loss": 0.3105,
+      "step": 120
+    },
+    {
+      "epoch": 0.38961038961038963,
+      "eval_loss": 0.31796982884407043,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9896,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 120
+    },
+    {
+      "epoch": 0.39285714285714285,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.6489849656263336e-05,
+      "loss": 0.3084,
+      "step": 121
+    },
+    {
+      "epoch": 0.3961038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.6396730215588913e-05,
+      "loss": 0.3124,
+      "step": 122
+    },
+    {
+      "epoch": 0.39935064935064934,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.630266232805709e-05,
+      "loss": 0.3167,
+      "step": 123
+    },
+    {
+      "epoch": 0.4025974025974026,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.6207659941167485e-05,
+      "loss": 0.3084,
+      "step": 124
+    },
+    {
+      "epoch": 0.40584415584415584,
+      "grad_norm": 0.62890625,
+      "learning_rate": 1.6111737140978495e-05,
+      "loss": 0.3143,
+      "step": 125
+    },
+    {
+      "epoch": 0.4090909090909091,
+      "grad_norm": 0.6171875,
+      "learning_rate": 1.6014908150018703e-05,
+      "loss": 0.3173,
+      "step": 126
+    },
+    {
+      "epoch": 0.41233766233766234,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.591718732517814e-05,
+      "loss": 0.3199,
+      "step": 127
+    },
+    {
+      "epoch": 0.4155844155844156,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.581858915557953e-05,
+      "loss": 0.312,
+      "step": 128
+    },
+    {
+      "epoch": 0.41883116883116883,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.571912826043003e-05,
+      "loss": 0.3217,
+      "step": 129
+    },
+    {
+      "epoch": 0.42207792207792205,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.5618819386853607e-05,
+      "loss": 0.3123,
+      "step": 130
+    },
+    {
+      "epoch": 0.42207792207792205,
+      "eval_loss": 0.3119741976261139,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9669,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 130
+    },
+    {
+      "epoch": 0.4253246753246753,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.551767740770446e-05,
+      "loss": 0.3088,
+      "step": 131
+    },
+    {
+      "epoch": 0.42857142857142855,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.541571731936185e-05,
+      "loss": 0.3084,
+      "step": 132
+    },
+    {
+      "epoch": 0.4318181818181818,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.5312954239506536e-05,
+      "loss": 0.3179,
+      "step": 133
+    },
+    {
+      "epoch": 0.43506493506493504,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.5209403404879305e-05,
+      "loss": 0.3145,
+      "step": 134
+    },
+    {
+      "epoch": 0.4383116883116883,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.5105080169021792e-05,
+      "loss": 0.3045,
+      "step": 135
+    },
+    {
+      "epoch": 0.44155844155844154,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.5000000000000002e-05,
+      "loss": 0.314,
+      "step": 136
+    },
+    {
+      "epoch": 0.4448051948051948,
+      "grad_norm": 0.6484375,
+      "learning_rate": 1.4894178478110856e-05,
+      "loss": 0.2985,
+      "step": 137
+    },
+    {
+      "epoch": 0.44805194805194803,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.4787631293572094e-05,
+      "loss": 0.3114,
+      "step": 138
+    },
+    {
+      "epoch": 0.4512987012987013,
+      "grad_norm": 0.60546875,
+      "learning_rate": 1.4680374244195861e-05,
+      "loss": 0.3129,
+      "step": 139
+    },
+    {
+      "epoch": 0.45454545454545453,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.4572423233046386e-05,
+      "loss": 0.3007,
+      "step": 140
+    },
+    {
+      "epoch": 0.45454545454545453,
+      "eval_loss": 0.30710509419441223,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9622,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 140
+    },
+    {
+      "epoch": 0.4577922077922078,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.4463794266081994e-05,
+      "loss": 0.2981,
+      "step": 141
+    },
+    {
+      "epoch": 0.461038961038961,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.4354503449781914e-05,
+      "loss": 0.2974,
+      "step": 142
+    },
+    {
+      "epoch": 0.4642857142857143,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.4244566988758152e-05,
+      "loss": 0.3064,
+      "step": 143
+    },
+    {
+      "epoch": 0.4675324675324675,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.4134001183352833e-05,
+      "loss": 0.298,
+      "step": 144
+    },
+    {
+      "epoch": 0.4707792207792208,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.4022822427221325e-05,
+      "loss": 0.3037,
+      "step": 145
+    },
+    {
+      "epoch": 0.474025974025974,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.391104720490156e-05,
+      "loss": 0.3062,
+      "step": 146
+    },
+    {
+      "epoch": 0.4772727272727273,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.3798692089369855e-05,
+      "loss": 0.3019,
+      "step": 147
+    },
+    {
+      "epoch": 0.4805194805194805,
+      "grad_norm": 0.61328125,
+      "learning_rate": 1.368577373958362e-05,
+      "loss": 0.3008,
+      "step": 148
+    },
+    {
+      "epoch": 0.4837662337662338,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.3572308898011328e-05,
+      "loss": 0.3055,
+      "step": 149
+    },
+    {
+      "epoch": 0.487012987012987,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.3458314388150115e-05,
+      "loss": 0.2942,
+      "step": 150
+    },
+    {
+      "epoch": 0.487012987012987,
+      "eval_loss": 0.30315953493118286,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9566,
+      "eval_samples_per_second": 11.883,
+      "eval_steps_per_second": 1.486,
+      "step": 150
+    },
+    {
+      "epoch": 0.4902597402597403,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.3343807112031329e-05,
+      "loss": 0.3097,
+      "step": 151
+    },
+    {
+      "epoch": 0.4935064935064935,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.3228804047714462e-05,
+      "loss": 0.3004,
+      "step": 152
+    },
+    {
+      "epoch": 0.4967532467532468,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.3113322246769817e-05,
+      "loss": 0.3046,
+      "step": 153
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.2997378831750242e-05,
+      "loss": 0.2966,
+      "step": 154
+    },
+    {
+      "epoch": 0.5032467532467533,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.2880990993652379e-05,
+      "loss": 0.2989,
+      "step": 155
+    },
+    {
+      "epoch": 0.5064935064935064,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.2764175989367717e-05,
+      "loss": 0.3015,
+      "step": 156
+    },
+    {
+      "epoch": 0.5097402597402597,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.2646951139123935e-05,
+      "loss": 0.2976,
+      "step": 157
+    },
+    {
+      "epoch": 0.512987012987013,
+      "grad_norm": 0.59765625,
+      "learning_rate": 1.2529333823916807e-05,
+      "loss": 0.2919,
+      "step": 158
+    },
+    {
+      "epoch": 0.5162337662337663,
+      "grad_norm": 0.6328125,
+      "learning_rate": 1.241134148293311e-05,
+      "loss": 0.3041,
+      "step": 159
+    },
+    {
+      "epoch": 0.5194805194805194,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.2292991610964902e-05,
+      "loss": 0.3053,
+      "step": 160
+    },
+    {
+      "epoch": 0.5194805194805194,
+      "eval_loss": 0.3000594675540924,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9643,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 160
+    },
+    {
+      "epoch": 0.5227272727272727,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.2174301755815572e-05,
+      "loss": 0.3043,
+      "step": 161
+    },
+    {
+      "epoch": 0.525974025974026,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.2055289515698008e-05,
+      "loss": 0.2988,
+      "step": 162
+    },
+    {
+      "epoch": 0.5292207792207793,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1935972536625302e-05,
+      "loss": 0.2871,
+      "step": 163
+    },
+    {
+      "epoch": 0.5324675324675324,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.1816368509794365e-05,
+      "loss": 0.2985,
+      "step": 164
+    },
+    {
+      "epoch": 0.5357142857142857,
+      "grad_norm": 0.609375,
+      "learning_rate": 1.1696495168962848e-05,
+      "loss": 0.3066,
+      "step": 165
+    },
+    {
+      "epoch": 0.538961038961039,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.1576370287819737e-05,
+      "loss": 0.296,
+      "step": 166
+    },
+    {
+      "epoch": 0.5422077922077922,
+      "grad_norm": 0.59375,
+      "learning_rate": 1.1456011677350052e-05,
+      "loss": 0.2997,
+      "step": 167
+    },
+    {
+      "epoch": 0.5454545454545454,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.133543718319398e-05,
+      "loss": 0.3011,
+      "step": 168
+    },
+    {
+      "epoch": 0.5487012987012987,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1214664683000927e-05,
+      "loss": 0.2898,
+      "step": 169
+    },
+    {
+      "epoch": 0.551948051948052,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.1093712083778748e-05,
+      "loss": 0.292,
+      "step": 170
+    },
+    {
+      "epoch": 0.551948051948052,
+      "eval_loss": 0.29717138409614563,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9869,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 170
+    },
+    {
+      "epoch": 0.5551948051948052,
+      "grad_norm": 0.578125,
+      "learning_rate": 1.0972597319238692e-05,
+      "loss": 0.2995,
+      "step": 171
+    },
+    {
+      "epoch": 0.5584415584415584,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.0851338347136358e-05,
+      "loss": 0.3015,
+      "step": 172
+    },
+    {
+      "epoch": 0.5616883116883117,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.0729953146609076e-05,
+      "loss": 0.2953,
+      "step": 173
+    },
+    {
+      "epoch": 0.564935064935065,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.060845971551014e-05,
+      "loss": 0.3011,
+      "step": 174
+    },
+    {
+      "epoch": 0.5681818181818182,
+      "grad_norm": 0.58984375,
+      "learning_rate": 1.0486876067740253e-05,
+      "loss": 0.3016,
+      "step": 175
+    },
+    {
+      "epoch": 0.5714285714285714,
+      "grad_norm": 0.5859375,
+      "learning_rate": 1.0365220230576592e-05,
+      "loss": 0.2953,
+      "step": 176
+    },
+    {
+      "epoch": 0.5746753246753247,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.0243510241999898e-05,
+      "loss": 0.2948,
+      "step": 177
+    },
+    {
+      "epoch": 0.577922077922078,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.0121764148019977e-05,
+      "loss": 0.2958,
+      "step": 178
+    },
+    {
+      "epoch": 0.5811688311688312,
+      "grad_norm": 0.5546875,
+      "learning_rate": 1e-05,
+      "loss": 0.2829,
+      "step": 179
+    },
+    {
+      "epoch": 0.5844155844155844,
+      "grad_norm": 0.58203125,
+      "learning_rate": 9.878235851980027e-06,
+      "loss": 0.2917,
+      "step": 180
+    },
+    {
+      "epoch": 0.5844155844155844,
+      "eval_loss": 0.29469791054725647,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9839,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 180
+    },
+    {
+      "epoch": 0.5876623376623377,
+      "grad_norm": 0.59375,
+      "learning_rate": 9.756489758000105e-06,
+      "loss": 0.2994,
+      "step": 181
+    },
+    {
+      "epoch": 0.5909090909090909,
+      "grad_norm": 0.57421875,
+      "learning_rate": 9.634779769423412e-06,
+      "loss": 0.3054,
+      "step": 182
+    },
+    {
+      "epoch": 0.5941558441558441,
+      "grad_norm": 0.59375,
+      "learning_rate": 9.51312393225975e-06,
+      "loss": 0.2925,
+      "step": 183
+    },
+    {
+      "epoch": 0.5974025974025974,
+      "grad_norm": 0.5703125,
+      "learning_rate": 9.391540284489862e-06,
+      "loss": 0.2816,
+      "step": 184
+    },
+    {
+      "epoch": 0.6006493506493507,
+      "grad_norm": 0.5390625,
+      "learning_rate": 9.270046853390924e-06,
+      "loss": 0.2856,
+      "step": 185
+    },
+    {
+      "epoch": 0.6038961038961039,
+      "grad_norm": 0.5703125,
+      "learning_rate": 9.148661652863644e-06,
+      "loss": 0.2936,
+      "step": 186
+    },
+    {
+      "epoch": 0.6071428571428571,
+      "grad_norm": 0.55078125,
+      "learning_rate": 9.02740268076131e-06,
+      "loss": 0.2938,
+      "step": 187
+    },
+    {
+      "epoch": 0.6103896103896104,
+      "grad_norm": 0.5859375,
+      "learning_rate": 8.906287916221259e-06,
+      "loss": 0.3017,
+      "step": 188
+    },
+    {
+      "epoch": 0.6136363636363636,
+      "grad_norm": 0.578125,
+      "learning_rate": 8.785335316999078e-06,
+      "loss": 0.2957,
+      "step": 189
+    },
+    {
+      "epoch": 0.6168831168831169,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.664562816806022e-06,
+      "loss": 0.2961,
+      "step": 190
+    },
+    {
+      "epoch": 0.6168831168831169,
+      "eval_loss": 0.2927069365978241,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9405,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 190
+    },
+    {
+      "epoch": 0.6201298701298701,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.543988322649954e-06,
+      "loss": 0.2915,
+      "step": 191
+    },
+    {
+      "epoch": 0.6233766233766234,
+      "grad_norm": 0.57421875,
+      "learning_rate": 8.423629712180265e-06,
+      "loss": 0.2875,
+      "step": 192
+    },
+    {
+      "epoch": 0.6266233766233766,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.303504831037154e-06,
+      "loss": 0.2979,
+      "step": 193
+    },
+    {
+      "epoch": 0.6298701298701299,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.183631490205636e-06,
+      "loss": 0.2891,
+      "step": 194
+    },
+    {
+      "epoch": 0.6331168831168831,
+      "grad_norm": 0.5703125,
+      "learning_rate": 8.064027463374702e-06,
+      "loss": 0.2796,
+      "step": 195
+    },
+    {
+      "epoch": 0.6363636363636364,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.944710484301995e-06,
+      "loss": 0.285,
+      "step": 196
+    },
+    {
+      "epoch": 0.6396103896103896,
+      "grad_norm": 0.58203125,
+      "learning_rate": 7.825698244184432e-06,
+      "loss": 0.295,
+      "step": 197
+    },
+    {
+      "epoch": 0.6428571428571429,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.707008389035102e-06,
+      "loss": 0.2962,
+      "step": 198
+    },
+    {
+      "epoch": 0.6461038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.588658517066893e-06,
+      "loss": 0.2912,
+      "step": 199
+    },
+    {
+      "epoch": 0.6493506493506493,
+      "grad_norm": 0.57421875,
+      "learning_rate": 7.470666176083193e-06,
+      "loss": 0.2917,
+      "step": 200
+    },
+    {
+      "epoch": 0.6493506493506493,
+      "eval_loss": 0.2912693917751312,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9875,
+      "eval_samples_per_second": 11.881,
+      "eval_steps_per_second": 1.486,
+      "step": 200
+    },
+    {
+      "epoch": 0.6525974025974026,
+      "grad_norm": 0.6015625,
+      "learning_rate": 7.3530488608760645e-06,
+      "loss": 0.291,
+      "step": 201
+    },
+    {
+      "epoch": 0.6558441558441559,
+      "grad_norm": 0.5625,
+      "learning_rate": 7.235824010632284e-06,
+      "loss": 0.2862,
+      "step": 202
+    },
+    {
+      "epoch": 0.6590909090909091,
+      "grad_norm": 0.5546875,
+      "learning_rate": 7.119009006347625e-06,
+      "loss": 0.2936,
+      "step": 203
+    },
+    {
+      "epoch": 0.6623376623376623,
+      "grad_norm": 0.5546875,
+      "learning_rate": 7.002621168249759e-06,
+      "loss": 0.2934,
+      "step": 204
+    },
+    {
+      "epoch": 0.6655844155844156,
+      "grad_norm": 0.578125,
+      "learning_rate": 6.886677753230184e-06,
+      "loss": 0.2783,
+      "step": 205
+    },
+    {
+      "epoch": 0.6688311688311688,
+      "grad_norm": 0.546875,
+      "learning_rate": 6.771195952285541e-06,
+      "loss": 0.2844,
+      "step": 206
+    },
+    {
+      "epoch": 0.672077922077922,
+      "grad_norm": 0.5546875,
+      "learning_rate": 6.656192887968675e-06,
+      "loss": 0.2835,
+      "step": 207
+    },
+    {
+      "epoch": 0.6753246753246753,
+      "grad_norm": 0.56640625,
+      "learning_rate": 6.5416856118498874e-06,
+      "loss": 0.2887,
+      "step": 208
+    },
+    {
+      "epoch": 0.6785714285714286,
+      "grad_norm": 0.5625,
+      "learning_rate": 6.427691101988673e-06,
+      "loss": 0.2925,
+      "step": 209
+    },
+    {
+      "epoch": 0.6818181818181818,
+      "grad_norm": 0.58203125,
+      "learning_rate": 6.314226260416383e-06,
+      "loss": 0.29,
+      "step": 210
+    },
+    {
+      "epoch": 0.6818181818181818,
+      "eval_loss": 0.29015374183654785,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.8645,
+      "eval_samples_per_second": 11.889,
+      "eval_steps_per_second": 1.487,
+      "step": 210
+    },
+    {
+      "epoch": 0.685064935064935,
+      "grad_norm": 0.5625,
+      "learning_rate": 6.2013079106301454e-06,
+      "loss": 0.2858,
+      "step": 211
+    },
+    {
+      "epoch": 0.6883116883116883,
+      "grad_norm": 0.5703125,
+      "learning_rate": 6.088952795098442e-06,
+      "loss": 0.2828,
+      "step": 212
+    },
+    {
+      "epoch": 0.6915584415584416,
+      "grad_norm": 0.578125,
+      "learning_rate": 5.977177572778679e-06,
+      "loss": 0.2947,
+      "step": 213
+    },
+    {
+      "epoch": 0.6948051948051948,
+      "grad_norm": 0.578125,
+      "learning_rate": 5.8659988166471715e-06,
+      "loss": 0.2795,
+      "step": 214
+    },
+    {
+      "epoch": 0.698051948051948,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.755433011241851e-06,
+      "loss": 0.2899,
+      "step": 215
+    },
+    {
+      "epoch": 0.7012987012987013,
+      "grad_norm": 0.578125,
+      "learning_rate": 5.645496550218089e-06,
+      "loss": 0.2881,
+      "step": 216
+    },
+    {
+      "epoch": 0.7045454545454546,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.5362057339180075e-06,
+      "loss": 0.2937,
+      "step": 217
+    },
+    {
+      "epoch": 0.7077922077922078,
+      "grad_norm": 0.5625,
+      "learning_rate": 5.427576766953615e-06,
+      "loss": 0.2828,
+      "step": 218
+    },
+    {
+      "epoch": 0.711038961038961,
+      "grad_norm": 0.57421875,
+      "learning_rate": 5.3196257558041386e-06,
+      "loss": 0.2863,
+      "step": 219
+    },
+    {
+      "epoch": 0.7142857142857143,
+      "grad_norm": 0.5859375,
+      "learning_rate": 5.212368706427913e-06,
+      "loss": 0.2876,
+      "step": 220
+    },
+    {
+      "epoch": 0.7142857142857143,
+      "eval_loss": 0.2891963720321655,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9738,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 220
+    },
+    {
+      "epoch": 0.7175324675324676,
+      "grad_norm": 0.5859375,
+      "learning_rate": 5.105821521889147e-06,
+      "loss": 0.2955,
+      "step": 221
+    },
+    {
+      "epoch": 0.7207792207792207,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.000000000000003e-06,
+      "loss": 0.2873,
+      "step": 222
+    },
+    {
+      "epoch": 0.724025974025974,
+      "grad_norm": 0.5625,
+      "learning_rate": 4.894919830978212e-06,
+      "loss": 0.2813,
+      "step": 223
+    },
+    {
+      "epoch": 0.7272727272727273,
+      "grad_norm": 0.59375,
+      "learning_rate": 4.790596595120699e-06,
+      "loss": 0.2912,
+      "step": 224
+    },
+    {
+      "epoch": 0.7305194805194806,
+      "grad_norm": 0.5859375,
+      "learning_rate": 4.687045760493468e-06,
+      "loss": 0.2921,
+      "step": 225
+    },
+    {
+      "epoch": 0.7337662337662337,
+      "grad_norm": 0.55078125,
+      "learning_rate": 4.584282680638155e-06,
+      "loss": 0.288,
+      "step": 226
+    },
+    {
+      "epoch": 0.737012987012987,
+      "grad_norm": 0.5546875,
+      "learning_rate": 4.482322592295541e-06,
+      "loss": 0.2914,
+      "step": 227
+    },
+    {
+      "epoch": 0.7402597402597403,
+      "grad_norm": 0.56640625,
+      "learning_rate": 4.381180613146396e-06,
+      "loss": 0.2901,
+      "step": 228
+    },
+    {
+      "epoch": 0.7435064935064936,
+      "grad_norm": 0.56640625,
+      "learning_rate": 4.280871739569972e-06,
+      "loss": 0.2941,
+      "step": 229
+    },
+    {
+      "epoch": 0.7467532467532467,
+      "grad_norm": 0.54296875,
+      "learning_rate": 4.181410844420473e-06,
+      "loss": 0.2836,
+      "step": 230
+    },
+    {
+      "epoch": 0.7467532467532467,
+      "eval_loss": 0.2885654866695404,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9437,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 230
+    },
+    {
+      "epoch": 0.75,
+      "grad_norm": 0.58203125,
+      "learning_rate": 4.082812674821865e-06,
+      "loss": 0.2882,
+      "step": 231
+    },
+    {
+      "epoch": 0.7532467532467533,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.9850918499812976e-06,
+      "loss": 0.293,
+      "step": 232
+    },
+    {
+      "epoch": 0.7564935064935064,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.888262859021508e-06,
+      "loss": 0.2818,
+      "step": 233
+    },
+    {
+      "epoch": 0.7597402597402597,
+      "grad_norm": 0.5703125,
+      "learning_rate": 3.7923400588325156e-06,
+      "loss": 0.2836,
+      "step": 234
+    },
+    {
+      "epoch": 0.762987012987013,
+      "grad_norm": 0.6015625,
+      "learning_rate": 3.6973376719429134e-06,
+      "loss": 0.2833,
+      "step": 235
+    },
+    {
+      "epoch": 0.7662337662337663,
+      "grad_norm": 0.5625,
+      "learning_rate": 3.6032697844110896e-06,
+      "loss": 0.2956,
+      "step": 236
+    },
+    {
+      "epoch": 0.7694805194805194,
+      "grad_norm": 0.578125,
+      "learning_rate": 3.5101503437366678e-06,
+      "loss": 0.288,
+      "step": 237
+    },
+    {
+      "epoch": 0.7727272727272727,
+      "grad_norm": 0.578125,
+      "learning_rate": 3.4179931567925216e-06,
+      "loss": 0.2898,
+      "step": 238
+    },
+    {
+      "epoch": 0.775974025974026,
+      "grad_norm": 0.55859375,
+      "learning_rate": 3.326811887777607e-06,
+      "loss": 0.2809,
+      "step": 239
+    },
+    {
+      "epoch": 0.7792207792207793,
+      "grad_norm": 0.546875,
+      "learning_rate": 3.236620056190972e-06,
+      "loss": 0.285,
+      "step": 240
+    },
+    {
+      "epoch": 0.7792207792207793,
+      "eval_loss": 0.2881048321723938,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9387,
+      "eval_samples_per_second": 11.884,
+      "eval_steps_per_second": 1.486,
+      "step": 240
+    },
+    {
+      "epoch": 0.7824675324675324,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.1474310348272084e-06,
+      "loss": 0.2888,
+      "step": 241
+    },
+    {
+      "epoch": 0.7857142857142857,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.0592580477936606e-06,
+      "loss": 0.2875,
+      "step": 242
+    },
+    {
+      "epoch": 0.788961038961039,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.9721141685496825e-06,
+      "loss": 0.2828,
+      "step": 243
+    },
+    {
+      "epoch": 0.7922077922077922,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.8860123179682244e-06,
+      "loss": 0.2866,
+      "step": 244
+    },
+    {
+      "epoch": 0.7954545454545454,
+      "grad_norm": 0.5703125,
+      "learning_rate": 2.8009652624200436e-06,
+      "loss": 0.289,
+      "step": 245
+    },
+    {
+      "epoch": 0.7987012987012987,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.7169856118808414e-06,
+      "loss": 0.2832,
+      "step": 246
+    },
+    {
+      "epoch": 0.801948051948052,
+      "grad_norm": 0.56640625,
+      "learning_rate": 2.634085818061565e-06,
+      "loss": 0.2913,
+      "step": 247
+    },
+    {
+      "epoch": 0.8051948051948052,
+      "grad_norm": 0.58203125,
+      "learning_rate": 2.5522781725621814e-06,
+      "loss": 0.2917,
+      "step": 248
+    },
+    {
+      "epoch": 0.8084415584415584,
+      "grad_norm": 0.56640625,
+      "learning_rate": 2.471574805049206e-06,
+      "loss": 0.2937,
+      "step": 249
+    },
+    {
+      "epoch": 0.8116883116883117,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.3919876814572197e-06,
+      "loss": 0.2932,
+      "step": 250
+    },
+    {
+      "epoch": 0.8116883116883117,
+      "eval_loss": 0.28785035014152527,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.993,
+      "eval_samples_per_second": 11.88,
+      "eval_steps_per_second": 1.486,
+      "step": 250
+    },
+    {
+      "epoch": 0.814935064935065,
+      "grad_norm": 0.56640625,
+      "learning_rate": 2.3135286022146785e-06,
+      "loss": 0.2973,
+      "step": 251
+    },
+    {
+      "epoch": 0.8181818181818182,
+      "grad_norm": 0.5703125,
+      "learning_rate": 2.2362092004942583e-06,
+      "loss": 0.2862,
+      "step": 252
+    },
+    {
+      "epoch": 0.8214285714285714,
+      "grad_norm": 0.57421875,
+      "learning_rate": 2.1600409404879875e-06,
+      "loss": 0.279,
+      "step": 253
+    },
+    {
+      "epoch": 0.8246753246753247,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.08503511570746e-06,
+      "loss": 0.2844,
+      "step": 254
+    },
+    {
+      "epoch": 0.827922077922078,
+      "grad_norm": 0.5546875,
+      "learning_rate": 2.0112028473093294e-06,
+      "loss": 0.2792,
+      "step": 255
+    },
+    {
+      "epoch": 0.8311688311688312,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.9385550824463727e-06,
+      "loss": 0.2921,
+      "step": 256
+    },
+    {
+      "epoch": 0.8344155844155844,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.8671025926443464e-06,
+      "loss": 0.2891,
+      "step": 257
+    },
+    {
+      "epoch": 0.8376623376623377,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.7968559722048906e-06,
+      "loss": 0.2848,
+      "step": 258
+    },
+    {
+      "epoch": 0.8409090909090909,
+      "grad_norm": 0.55078125,
+      "learning_rate": 1.7278256366347034e-06,
+      "loss": 0.2921,
+      "step": 259
+    },
+    {
+      "epoch": 0.8441558441558441,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.660021821101222e-06,
+      "loss": 0.2907,
+      "step": 260
+    },
+    {
+      "epoch": 0.8441558441558441,
+      "eval_loss": 0.2877131402492523,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.008,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 260
+    },
+    {
+      "epoch": 0.8474025974025974,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.5934545789150625e-06,
+      "loss": 0.2979,
+      "step": 261
+    },
+    {
+      "epoch": 0.8506493506493507,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.528133780039397e-06,
+      "loss": 0.2847,
+      "step": 262
+    },
+    {
+      "epoch": 0.8538961038961039,
+      "grad_norm": 0.55078125,
+      "learning_rate": 1.4640691096265358e-06,
+      "loss": 0.2902,
+      "step": 263
+    },
+    {
+      "epoch": 0.8571428571428571,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.401270066581899e-06,
+      "loss": 0.2861,
+      "step": 264
+    },
+    {
+      "epoch": 0.8603896103896104,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.339745962155613e-06,
+      "loss": 0.2823,
+      "step": 265
+    },
+    {
+      "epoch": 0.8636363636363636,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.279505918561923e-06,
+      "loss": 0.2855,
+      "step": 266
+    },
+    {
+      "epoch": 0.8668831168831169,
+      "grad_norm": 0.58203125,
+      "learning_rate": 1.220558867626639e-06,
+      "loss": 0.2853,
+      "step": 267
+    },
+    {
+      "epoch": 0.8701298701298701,
+      "grad_norm": 0.5703125,
+      "learning_rate": 1.1629135494628097e-06,
+      "loss": 0.2891,
+      "step": 268
+    },
+    {
+      "epoch": 0.8733766233766234,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.1065785111748117e-06,
+      "loss": 0.2853,
+      "step": 269
+    },
+    {
+      "epoch": 0.8766233766233766,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.051562105591082e-06,
+      "loss": 0.292,
+      "step": 270
+    },
+    {
+      "epoch": 0.8766233766233766,
+      "eval_loss": 0.28765836358070374,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.932,
+      "eval_samples_per_second": 11.885,
+      "eval_steps_per_second": 1.486,
+      "step": 270
+    },
+    {
+      "epoch": 0.8798701298701299,
+      "grad_norm": 0.55859375,
+      "learning_rate": 9.978724900256265e-07,
+      "loss": 0.2825,
+      "step": 271
+    },
+    {
+      "epoch": 0.8831168831168831,
+      "grad_norm": 0.5546875,
+      "learning_rate": 9.455176250685338e-07,
+      "loss": 0.2836,
+      "step": 272
+    },
+    {
+      "epoch": 0.8863636363636364,
+      "grad_norm": 0.58203125,
+      "learning_rate": 8.945052734056581e-07,
+      "loss": 0.2959,
+      "step": 273
+    },
+    {
+      "epoch": 0.8896103896103896,
+      "grad_norm": 0.57421875,
+      "learning_rate": 8.448429986676298e-07,
+      "loss": 0.3003,
+      "step": 274
+    },
+    {
+      "epoch": 0.8928571428571429,
+      "grad_norm": 0.56640625,
+      "learning_rate": 7.965381643084069e-07,
+      "loss": 0.295,
+      "step": 275
+    },
+    {
+      "epoch": 0.8961038961038961,
+      "grad_norm": 0.5625,
+      "learning_rate": 7.495979325134806e-07,
+      "loss": 0.2873,
+      "step": 276
+    },
+    {
+      "epoch": 0.8993506493506493,
+      "grad_norm": 0.55078125,
+      "learning_rate": 7.040292631379386e-07,
+      "loss": 0.2827,
+      "step": 277
+    },
+    {
+      "epoch": 0.9025974025974026,
+      "grad_norm": 0.54296875,
+      "learning_rate": 6.598389126745209e-07,
+      "loss": 0.2797,
+      "step": 278
+    },
+    {
+      "epoch": 0.9058441558441559,
+      "grad_norm": 0.53125,
+      "learning_rate": 6.170334332518325e-07,
+      "loss": 0.272,
+      "step": 279
+    },
+    {
+      "epoch": 0.9090909090909091,
+      "grad_norm": 0.58203125,
+      "learning_rate": 5.756191716628556e-07,
+      "loss": 0.292,
+      "step": 280
+    },
+    {
+      "epoch": 0.9090909090909091,
+      "eval_loss": 0.28761279582977295,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9741,
+      "eval_samples_per_second": 11.882,
+      "eval_steps_per_second": 1.486,
+      "step": 280
+    },
+    {
+      "epoch": 0.9123376623376623,
+      "grad_norm": 0.5703125,
+      "learning_rate": 5.35602268423906e-07,
+      "loss": 0.2961,
+      "step": 281
+    },
+    {
+      "epoch": 0.9155844155844156,
+      "grad_norm": 0.578125,
+      "learning_rate": 4.969886568641757e-07,
+      "loss": 0.2904,
+      "step": 282
+    },
+    {
+      "epoch": 0.9188311688311688,
+      "grad_norm": 0.5625,
+      "learning_rate": 4.597840622459937e-07,
+      "loss": 0.2874,
+      "step": 283
+    },
+    {
+      "epoch": 0.922077922077922,
+      "grad_norm": 0.546875,
+      "learning_rate": 4.2399400091594154e-07,
+      "loss": 0.2794,
+      "step": 284
+    },
+    {
+      "epoch": 0.9253246753246753,
+      "grad_norm": 0.578125,
+      "learning_rate": 3.8962377948693395e-07,
+      "loss": 0.2962,
+      "step": 285
+    },
+    {
+      "epoch": 0.9285714285714286,
+      "grad_norm": 0.56640625,
+      "learning_rate": 3.566784940514145e-07,
+      "loss": 0.2914,
+      "step": 286
+    },
+    {
+      "epoch": 0.9318181818181818,
+      "grad_norm": 0.5703125,
+      "learning_rate": 3.2516302942574794e-07,
+      "loss": 0.2838,
+      "step": 287
+    },
+    {
+      "epoch": 0.935064935064935,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.9508205842594727e-07,
+      "loss": 0.2812,
+      "step": 288
+    },
+    {
+      "epoch": 0.9383116883116883,
+      "grad_norm": 0.5625,
+      "learning_rate": 2.6644004117483357e-07,
+      "loss": 0.2827,
+      "step": 289
+    },
+    {
+      "epoch": 0.9415584415584416,
+      "grad_norm": 0.5546875,
+      "learning_rate": 2.392412244407294e-07,
+      "loss": 0.2791,
+      "step": 290
+    },
+    {
+      "epoch": 0.9415584415584416,
+      "eval_loss": 0.28759849071502686,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 175.0075,
+      "eval_samples_per_second": 11.879,
+      "eval_steps_per_second": 1.486,
+      "step": 290
+    },
+    {
+      "epoch": 0.9448051948051948,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.1348964100778914e-07,
+      "loss": 0.285,
+      "step": 291
+    },
+    {
+      "epoch": 0.948051948051948,
+      "grad_norm": 0.55859375,
+      "learning_rate": 1.8918910907805733e-07,
+      "loss": 0.288,
+      "step": 292
+    },
+    {
+      "epoch": 0.9512987012987013,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.6634323170533928e-07,
+      "loss": 0.2854,
+      "step": 293
+    },
+    {
+      "epoch": 0.9545454545454546,
+      "grad_norm": 0.57421875,
+      "learning_rate": 1.4495539626097289e-07,
+      "loss": 0.2873,
+      "step": 294
+    },
+    {
+      "epoch": 0.9577922077922078,
+      "grad_norm": 0.56640625,
+      "learning_rate": 1.2502877393158587e-07,
+      "loss": 0.2841,
+      "step": 295
+    },
+    {
+      "epoch": 0.961038961038961,
+      "grad_norm": 0.62109375,
+      "learning_rate": 1.0656631924889749e-07,
+      "loss": 0.2994,
+      "step": 296
+    },
+    {
+      "epoch": 0.9642857142857143,
+      "grad_norm": 0.5546875,
+      "learning_rate": 8.957076965165234e-08,
+      "loss": 0.2797,
+      "step": 297
+    },
+    {
+      "epoch": 0.9675324675324676,
+      "grad_norm": 0.56640625,
+      "learning_rate": 7.404464507973608e-08,
+      "loss": 0.2868,
+      "step": 298
+    },
+    {
+      "epoch": 0.9707792207792207,
+      "grad_norm": 0.56640625,
+      "learning_rate": 5.999024760054095e-08,
+      "loss": 0.2867,
+      "step": 299
+    },
+    {
+      "epoch": 0.974025974025974,
+      "grad_norm": 0.56640625,
+      "learning_rate": 4.740966106764222e-08,
+      "loss": 0.2832,
+      "step": 300
+    },
+    {
+      "epoch": 0.974025974025974,
+      "eval_loss": 0.287624329328537,
+      "eval_model_preparation_time": 0.0003,
+      "eval_runtime": 174.9045,
+      "eval_samples_per_second": 11.886,
+      "eval_steps_per_second": 1.487,
+      "step": 300
+    },
+    {
+      "epoch": 0.9772727272727273,
+      "grad_norm": 0.5625,
+      "learning_rate": 3.630475081181861e-08,
+      "loss": 0.2818,
+      "step": 301
+    },
+    {
+      "epoch": 0.9805194805194806,
+      "grad_norm": 0.57421875,
+      "learning_rate": 2.667716336448356e-08,
+      "loss": 0.2874,
+      "step": 302
+    },
+    {
+      "epoch": 0.9837662337662337,
+      "grad_norm": 0.5546875,
+      "learning_rate": 1.8528326213548276e-08,
+      "loss": 0.2839,
+      "step": 303
+    },
+    {
+      "epoch": 0.987012987012987,
+      "grad_norm": 0.5625,
+      "learning_rate": 1.1859447591769934e-08,
+      "loss": 0.2911,
+      "step": 304
+    },
+    {
+      "epoch": 0.9902597402597403,
+      "grad_norm": 0.59765625,
+      "learning_rate": 6.671516297606095e-09,
+      "loss": 0.2934,
+      "step": 305
+    },
+    {
+      "epoch": 0.9935064935064936,
+      "grad_norm": 0.55859375,
+      "learning_rate": 2.9653015486064143e-09,
+      "loss": 0.287,
+      "step": 306
+    },
+    {
+      "epoch": 0.9967532467532467,
+      "grad_norm": 0.5859375,
+      "learning_rate": 7.413528673549941e-10,
+      "loss": 0.2909,
+      "step": 307
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 0.58984375,
+      "learning_rate": 0.0,
+      "loss": 0.2824,
+      "step": 308
+    }
+  ],
+  "logging_steps": 1,
+  "max_steps": 308,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 100,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 1.8176663382235546e+18,
+  "train_batch_size": 8,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/checkpoint-308/training_args.bin b/checkpoint-308/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..4684d3c90558d6bfde405c82b58596bc6231bfee
--- /dev/null
+++ b/checkpoint-308/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d09f69be12f7ef607f0a1508435a517f5277e3160e5d6a59346e619fcbaa3592
+size 5176
diff --git a/config.json b/config.json
new file mode 100644
index 0000000000000000000000000000000000000000..de55730a672c2805e1108109d81a111b0cdb48b1
--- /dev/null
+++ b/config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f8b621112eddc9da772fae3f70d672f4d949cc5cb1d580f135d66b3836598c1a
+size 915
diff --git a/generation_config.json b/generation_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..1007819213eaa3e36dd0c6354016734419f4a863
--- /dev/null
+++ b/generation_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a7327788370b7783e29852cf19b4a1f40cd31af5962c94bba76c336d491b4169
+size 184
diff --git a/model-00001-of-00004.safetensors b/model-00001-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..616bb67b1f5310e3c9955e638763bdeff0fb55c8
--- /dev/null
+++ b/model-00001-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3abfe37c12d3907a891b3b7282c09590bedc9a560e05339e0aebfba26341dc1b
+size 4976698672
diff --git a/model-00002-of-00004.safetensors b/model-00002-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..4c1a59736868aa002e16da3948c4b68be6d97a71
--- /dev/null
+++ b/model-00002-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:584021d05e0ffd6879fcb1af88d896f418db0a6656a6c799d17e6a28e6540c8b
+size 4999802720
diff --git a/model-00003-of-00004.safetensors b/model-00003-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..075dada5dcfe243936f5583203b90ba80f0935e1
--- /dev/null
+++ b/model-00003-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ced290d33c8beefb926d706c895d70bd58e0d89ba90daca1d637367a6e9163ae
+size 4915916176
diff --git a/model-00004-of-00004.safetensors b/model-00004-of-00004.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..b1e38fdf23d6ff664ce00189e3dc1ab419ed8545
--- /dev/null
+++ b/model-00004-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:03dad76ab9974b3c84fcd8067e459ff22c49e10e1408ea16637a06df0b9b71c4
+size 1168138808
diff --git a/model.safetensors.index.json b/model.safetensors.index.json
new file mode 100644
index 0000000000000000000000000000000000000000..a054aad8cf0cb5671e2e10b3bf817725a4bec031
--- /dev/null
+++ b/model.safetensors.index.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:146776fce3f6db1103aa6f249e65ee5544c5923ce6f971b092eee79aa6e5d37b
+size 23950
diff --git a/special_tokens_map.json b/special_tokens_map.json
new file mode 100644
index 0000000000000000000000000000000000000000..92d7fab6729d723f45689803b99006fec945cc47
--- /dev/null
+++ b/special_tokens_map.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b1835caa5b4d70acaa210fa222b0036f1882f9525c4660fd4810fb3e1e40ff8
+size 325
diff --git a/tokenizer.json b/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..66cd9d7e0daec95eb10d16a63c615637dbbb7304
--- /dev/null
+++ b/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79e3e522635f3171300913bb421464a87de6222182a0570b9b2ccba2a964b2b4
+size 9085657
diff --git a/tokenizer_config.json b/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..08ebc9598183a9a76c44ef06813aca75f51de4a5
--- /dev/null
+++ b/tokenizer_config.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:05d0de6d831b5940c3f2eae4e495b9f77612cbb05b5804846733ab617f0dfe9b
+size 55380