{ "best_metric": 0.9823666485038896, "best_model_checkpoint": "FFPP-Raw_1FPS_faces-expand-40-aligned\\checkpoint-4044", "epoch": 3.0, "eval_steps": 500, "global_step": 4044, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.01, "grad_norm": 8.450878143310547, "learning_rate": 1.234567901234568e-06, "loss": 0.7645, "step": 10 }, { "epoch": 0.01, "grad_norm": 7.411558628082275, "learning_rate": 2.469135802469136e-06, "loss": 0.6817, "step": 20 }, { "epoch": 0.02, "grad_norm": 4.373965740203857, "learning_rate": 3.7037037037037037e-06, "loss": 0.5998, "step": 30 }, { "epoch": 0.03, "grad_norm": 2.776858329772949, "learning_rate": 4.938271604938272e-06, "loss": 0.5307, "step": 40 }, { "epoch": 0.04, "grad_norm": 5.085646629333496, "learning_rate": 6.172839506172839e-06, "loss": 0.525, "step": 50 }, { "epoch": 0.04, "grad_norm": 3.5710902214050293, "learning_rate": 7.4074074074074075e-06, "loss": 0.4941, "step": 60 }, { "epoch": 0.05, "grad_norm": 2.6829416751861572, "learning_rate": 8.641975308641975e-06, "loss": 0.5426, "step": 70 }, { "epoch": 0.06, "grad_norm": 3.156679630279541, "learning_rate": 9.876543209876543e-06, "loss": 0.4969, "step": 80 }, { "epoch": 0.07, "grad_norm": 2.465975284576416, "learning_rate": 1.1111111111111112e-05, "loss": 0.5458, "step": 90 }, { "epoch": 0.07, "grad_norm": 2.412788152694702, "learning_rate": 1.2345679012345678e-05, "loss": 0.5162, "step": 100 }, { "epoch": 0.08, "grad_norm": 2.6743991374969482, "learning_rate": 1.3580246913580247e-05, "loss": 0.5121, "step": 110 }, { "epoch": 0.09, "grad_norm": 4.343439102172852, "learning_rate": 1.4814814814814815e-05, "loss": 0.5369, "step": 120 }, { "epoch": 0.1, "grad_norm": 4.486733913421631, "learning_rate": 1.604938271604938e-05, "loss": 0.5169, "step": 130 }, { "epoch": 0.1, "grad_norm": 4.967243194580078, "learning_rate": 1.728395061728395e-05, "loss": 0.5097, "step": 140 }, { "epoch": 0.11, "grad_norm": 3.3753137588500977, "learning_rate": 1.8518518518518518e-05, "loss": 0.4909, "step": 150 }, { "epoch": 0.12, "grad_norm": 5.698143005371094, "learning_rate": 1.9753086419753087e-05, "loss": 0.5047, "step": 160 }, { "epoch": 0.13, "grad_norm": 4.35822057723999, "learning_rate": 2.0987654320987655e-05, "loss": 0.5101, "step": 170 }, { "epoch": 0.13, "grad_norm": 4.044003009796143, "learning_rate": 2.2222222222222223e-05, "loss": 0.4809, "step": 180 }, { "epoch": 0.14, "grad_norm": 5.79054594039917, "learning_rate": 2.345679012345679e-05, "loss": 0.5084, "step": 190 }, { "epoch": 0.15, "grad_norm": 5.470571041107178, "learning_rate": 2.4691358024691357e-05, "loss": 0.4427, "step": 200 }, { "epoch": 0.16, "grad_norm": 7.30893611907959, "learning_rate": 2.5925925925925925e-05, "loss": 0.4598, "step": 210 }, { "epoch": 0.16, "grad_norm": 10.693650245666504, "learning_rate": 2.7160493827160493e-05, "loss": 0.4888, "step": 220 }, { "epoch": 0.17, "grad_norm": 6.367429733276367, "learning_rate": 2.839506172839506e-05, "loss": 0.4479, "step": 230 }, { "epoch": 0.18, "grad_norm": 4.16640567779541, "learning_rate": 2.962962962962963e-05, "loss": 0.431, "step": 240 }, { "epoch": 0.19, "grad_norm": 11.808082580566406, "learning_rate": 3.08641975308642e-05, "loss": 0.4055, "step": 250 }, { "epoch": 0.19, "grad_norm": 21.1939640045166, "learning_rate": 3.209876543209876e-05, "loss": 0.3949, "step": 260 }, { "epoch": 0.2, "grad_norm": 8.935243606567383, "learning_rate": 3.3333333333333335e-05, "loss": 0.3778, "step": 270 }, { "epoch": 0.21, "grad_norm": 7.449581146240234, "learning_rate": 3.45679012345679e-05, "loss": 0.3449, "step": 280 }, { "epoch": 0.22, "grad_norm": 8.583562850952148, "learning_rate": 3.580246913580247e-05, "loss": 0.3957, "step": 290 }, { "epoch": 0.22, "grad_norm": 6.4094133377075195, "learning_rate": 3.7037037037037037e-05, "loss": 0.3931, "step": 300 }, { "epoch": 0.23, "grad_norm": 7.312207221984863, "learning_rate": 3.82716049382716e-05, "loss": 0.3241, "step": 310 }, { "epoch": 0.24, "grad_norm": 14.610054969787598, "learning_rate": 3.950617283950617e-05, "loss": 0.3262, "step": 320 }, { "epoch": 0.24, "grad_norm": 11.481966018676758, "learning_rate": 4.074074074074074e-05, "loss": 0.3301, "step": 330 }, { "epoch": 0.25, "grad_norm": 14.101536750793457, "learning_rate": 4.197530864197531e-05, "loss": 0.3068, "step": 340 }, { "epoch": 0.26, "grad_norm": 7.435379981994629, "learning_rate": 4.3209876543209875e-05, "loss": 0.413, "step": 350 }, { "epoch": 0.27, "grad_norm": 26.36928939819336, "learning_rate": 4.4444444444444447e-05, "loss": 0.3646, "step": 360 }, { "epoch": 0.27, "grad_norm": 7.532124996185303, "learning_rate": 4.567901234567901e-05, "loss": 0.4034, "step": 370 }, { "epoch": 0.28, "grad_norm": 9.976860046386719, "learning_rate": 4.691358024691358e-05, "loss": 0.3523, "step": 380 }, { "epoch": 0.29, "grad_norm": 11.547489166259766, "learning_rate": 4.814814814814815e-05, "loss": 0.3585, "step": 390 }, { "epoch": 0.3, "grad_norm": 9.796356201171875, "learning_rate": 4.938271604938271e-05, "loss": 0.3474, "step": 400 }, { "epoch": 0.3, "grad_norm": 7.1311421394348145, "learning_rate": 4.993129980763946e-05, "loss": 0.3042, "step": 410 }, { "epoch": 0.31, "grad_norm": 18.114513397216797, "learning_rate": 4.979389942291839e-05, "loss": 0.3492, "step": 420 }, { "epoch": 0.32, "grad_norm": 14.068829536437988, "learning_rate": 4.965649903819731e-05, "loss": 0.2967, "step": 430 }, { "epoch": 0.33, "grad_norm": 22.513200759887695, "learning_rate": 4.951909865347623e-05, "loss": 0.3142, "step": 440 }, { "epoch": 0.33, "grad_norm": 17.25227165222168, "learning_rate": 4.9381698268755155e-05, "loss": 0.3429, "step": 450 }, { "epoch": 0.34, "grad_norm": 7.3096699714660645, "learning_rate": 4.924429788403408e-05, "loss": 0.2944, "step": 460 }, { "epoch": 0.35, "grad_norm": 7.90877628326416, "learning_rate": 4.9106897499313e-05, "loss": 0.3029, "step": 470 }, { "epoch": 0.36, "grad_norm": 14.439558982849121, "learning_rate": 4.896949711459192e-05, "loss": 0.2862, "step": 480 }, { "epoch": 0.36, "grad_norm": 7.571442127227783, "learning_rate": 4.883209672987085e-05, "loss": 0.3197, "step": 490 }, { "epoch": 0.37, "grad_norm": 9.224518775939941, "learning_rate": 4.8694696345149774e-05, "loss": 0.3173, "step": 500 }, { "epoch": 0.38, "grad_norm": 19.048572540283203, "learning_rate": 4.8557295960428687e-05, "loss": 0.3749, "step": 510 }, { "epoch": 0.39, "grad_norm": 7.699029922485352, "learning_rate": 4.841989557570761e-05, "loss": 0.3774, "step": 520 }, { "epoch": 0.39, "grad_norm": 10.619318008422852, "learning_rate": 4.828249519098654e-05, "loss": 0.3161, "step": 530 }, { "epoch": 0.4, "grad_norm": 12.99872875213623, "learning_rate": 4.814509480626546e-05, "loss": 0.2833, "step": 540 }, { "epoch": 0.41, "grad_norm": 12.019877433776855, "learning_rate": 4.800769442154438e-05, "loss": 0.2371, "step": 550 }, { "epoch": 0.42, "grad_norm": 17.415578842163086, "learning_rate": 4.7870294036823306e-05, "loss": 0.2497, "step": 560 }, { "epoch": 0.42, "grad_norm": 10.935409545898438, "learning_rate": 4.7732893652102225e-05, "loss": 0.2674, "step": 570 }, { "epoch": 0.43, "grad_norm": 7.685445785522461, "learning_rate": 4.759549326738115e-05, "loss": 0.3019, "step": 580 }, { "epoch": 0.44, "grad_norm": 6.198949337005615, "learning_rate": 4.745809288266007e-05, "loss": 0.2686, "step": 590 }, { "epoch": 0.45, "grad_norm": 14.602446556091309, "learning_rate": 4.7320692497939e-05, "loss": 0.2582, "step": 600 }, { "epoch": 0.45, "grad_norm": 23.308094024658203, "learning_rate": 4.718329211321792e-05, "loss": 0.3482, "step": 610 }, { "epoch": 0.46, "grad_norm": 19.93767547607422, "learning_rate": 4.7045891728496844e-05, "loss": 0.2962, "step": 620 }, { "epoch": 0.47, "grad_norm": 15.824058532714844, "learning_rate": 4.6908491343775764e-05, "loss": 0.2796, "step": 630 }, { "epoch": 0.47, "grad_norm": 7.9304609298706055, "learning_rate": 4.6771090959054684e-05, "loss": 0.2712, "step": 640 }, { "epoch": 0.48, "grad_norm": 14.579902648925781, "learning_rate": 4.663369057433361e-05, "loss": 0.2783, "step": 650 }, { "epoch": 0.49, "grad_norm": 9.262262344360352, "learning_rate": 4.649629018961254e-05, "loss": 0.2541, "step": 660 }, { "epoch": 0.5, "grad_norm": 8.660795211791992, "learning_rate": 4.6358889804891456e-05, "loss": 0.2636, "step": 670 }, { "epoch": 0.5, "grad_norm": 24.701696395874023, "learning_rate": 4.6221489420170376e-05, "loss": 0.3344, "step": 680 }, { "epoch": 0.51, "grad_norm": 10.420050621032715, "learning_rate": 4.60840890354493e-05, "loss": 0.3001, "step": 690 }, { "epoch": 0.52, "grad_norm": 18.686073303222656, "learning_rate": 4.594668865072822e-05, "loss": 0.2415, "step": 700 }, { "epoch": 0.53, "grad_norm": 13.247358322143555, "learning_rate": 4.580928826600714e-05, "loss": 0.3413, "step": 710 }, { "epoch": 0.53, "grad_norm": 9.456878662109375, "learning_rate": 4.567188788128607e-05, "loss": 0.2764, "step": 720 }, { "epoch": 0.54, "grad_norm": 13.153331756591797, "learning_rate": 4.5534487496564995e-05, "loss": 0.2365, "step": 730 }, { "epoch": 0.55, "grad_norm": 9.744629859924316, "learning_rate": 4.5397087111843915e-05, "loss": 0.2239, "step": 740 }, { "epoch": 0.56, "grad_norm": 10.051612854003906, "learning_rate": 4.525968672712284e-05, "loss": 0.2357, "step": 750 }, { "epoch": 0.56, "grad_norm": 7.455211639404297, "learning_rate": 4.512228634240176e-05, "loss": 0.2401, "step": 760 }, { "epoch": 0.57, "grad_norm": 14.734637260437012, "learning_rate": 4.498488595768068e-05, "loss": 0.2987, "step": 770 }, { "epoch": 0.58, "grad_norm": 13.91230583190918, "learning_rate": 4.484748557295961e-05, "loss": 0.1869, "step": 780 }, { "epoch": 0.59, "grad_norm": 11.488323211669922, "learning_rate": 4.4710085188238534e-05, "loss": 0.2134, "step": 790 }, { "epoch": 0.59, "grad_norm": 7.693485736846924, "learning_rate": 4.4572684803517453e-05, "loss": 0.2026, "step": 800 }, { "epoch": 0.6, "grad_norm": 6.261735916137695, "learning_rate": 4.443528441879637e-05, "loss": 0.2343, "step": 810 }, { "epoch": 0.61, "grad_norm": 6.214202404022217, "learning_rate": 4.42978840340753e-05, "loss": 0.2183, "step": 820 }, { "epoch": 0.62, "grad_norm": 7.414583683013916, "learning_rate": 4.4160483649354226e-05, "loss": 0.1992, "step": 830 }, { "epoch": 0.62, "grad_norm": 13.354372024536133, "learning_rate": 4.402308326463314e-05, "loss": 0.2622, "step": 840 }, { "epoch": 0.63, "grad_norm": 13.756591796875, "learning_rate": 4.3885682879912066e-05, "loss": 0.2207, "step": 850 }, { "epoch": 0.64, "grad_norm": 16.665782928466797, "learning_rate": 4.374828249519099e-05, "loss": 0.2232, "step": 860 }, { "epoch": 0.65, "grad_norm": 9.784102439880371, "learning_rate": 4.361088211046991e-05, "loss": 0.2174, "step": 870 }, { "epoch": 0.65, "grad_norm": 9.463679313659668, "learning_rate": 4.347348172574883e-05, "loss": 0.2443, "step": 880 }, { "epoch": 0.66, "grad_norm": 12.118477821350098, "learning_rate": 4.333608134102776e-05, "loss": 0.2259, "step": 890 }, { "epoch": 0.67, "grad_norm": 10.19271183013916, "learning_rate": 4.319868095630668e-05, "loss": 0.2127, "step": 900 }, { "epoch": 0.68, "grad_norm": 14.1488037109375, "learning_rate": 4.3061280571585604e-05, "loss": 0.2007, "step": 910 }, { "epoch": 0.68, "grad_norm": 6.0573320388793945, "learning_rate": 4.2923880186864524e-05, "loss": 0.2268, "step": 920 }, { "epoch": 0.69, "grad_norm": 7.142855644226074, "learning_rate": 4.278647980214345e-05, "loss": 0.2263, "step": 930 }, { "epoch": 0.7, "grad_norm": 9.926745414733887, "learning_rate": 4.264907941742237e-05, "loss": 0.2133, "step": 940 }, { "epoch": 0.7, "grad_norm": 12.452362060546875, "learning_rate": 4.25116790327013e-05, "loss": 0.2, "step": 950 }, { "epoch": 0.71, "grad_norm": 7.340023994445801, "learning_rate": 4.2374278647980216e-05, "loss": 0.1764, "step": 960 }, { "epoch": 0.72, "grad_norm": 8.337459564208984, "learning_rate": 4.2236878263259136e-05, "loss": 0.2465, "step": 970 }, { "epoch": 0.73, "grad_norm": 15.628137588500977, "learning_rate": 4.209947787853806e-05, "loss": 0.2005, "step": 980 }, { "epoch": 0.73, "grad_norm": 5.059557914733887, "learning_rate": 4.196207749381699e-05, "loss": 0.2389, "step": 990 }, { "epoch": 0.74, "grad_norm": 7.46613073348999, "learning_rate": 4.18246771090959e-05, "loss": 0.2134, "step": 1000 }, { "epoch": 0.75, "grad_norm": 7.257927894592285, "learning_rate": 4.168727672437483e-05, "loss": 0.1958, "step": 1010 }, { "epoch": 0.76, "grad_norm": 9.179327011108398, "learning_rate": 4.1549876339653755e-05, "loss": 0.1929, "step": 1020 }, { "epoch": 0.76, "grad_norm": 22.463008880615234, "learning_rate": 4.1412475954932675e-05, "loss": 0.2206, "step": 1030 }, { "epoch": 0.77, "grad_norm": 6.543388366699219, "learning_rate": 4.1275075570211595e-05, "loss": 0.1832, "step": 1040 }, { "epoch": 0.78, "grad_norm": 6.427690029144287, "learning_rate": 4.113767518549052e-05, "loss": 0.1845, "step": 1050 }, { "epoch": 0.79, "grad_norm": 10.30201244354248, "learning_rate": 4.100027480076945e-05, "loss": 0.2071, "step": 1060 }, { "epoch": 0.79, "grad_norm": 8.301946640014648, "learning_rate": 4.086287441604837e-05, "loss": 0.2033, "step": 1070 }, { "epoch": 0.8, "grad_norm": 12.093053817749023, "learning_rate": 4.072547403132729e-05, "loss": 0.1897, "step": 1080 }, { "epoch": 0.81, "grad_norm": 11.62564754486084, "learning_rate": 4.0588073646606214e-05, "loss": 0.1791, "step": 1090 }, { "epoch": 0.82, "grad_norm": 9.241898536682129, "learning_rate": 4.045067326188513e-05, "loss": 0.2097, "step": 1100 }, { "epoch": 0.82, "grad_norm": 6.859925746917725, "learning_rate": 4.031327287716406e-05, "loss": 0.192, "step": 1110 }, { "epoch": 0.83, "grad_norm": 13.545909881591797, "learning_rate": 4.017587249244298e-05, "loss": 0.2016, "step": 1120 }, { "epoch": 0.84, "grad_norm": 9.510741233825684, "learning_rate": 4.0038472107721906e-05, "loss": 0.2266, "step": 1130 }, { "epoch": 0.85, "grad_norm": 12.568158149719238, "learning_rate": 3.9901071723000826e-05, "loss": 0.2498, "step": 1140 }, { "epoch": 0.85, "grad_norm": 14.593520164489746, "learning_rate": 3.976367133827975e-05, "loss": 0.1981, "step": 1150 }, { "epoch": 0.86, "grad_norm": 5.609714508056641, "learning_rate": 3.962627095355867e-05, "loss": 0.1987, "step": 1160 }, { "epoch": 0.87, "grad_norm": 8.593751907348633, "learning_rate": 3.948887056883759e-05, "loss": 0.1862, "step": 1170 }, { "epoch": 0.88, "grad_norm": 5.881361961364746, "learning_rate": 3.935147018411652e-05, "loss": 0.2006, "step": 1180 }, { "epoch": 0.88, "grad_norm": 9.768789291381836, "learning_rate": 3.9214069799395445e-05, "loss": 0.2235, "step": 1190 }, { "epoch": 0.89, "grad_norm": 7.065619468688965, "learning_rate": 3.907666941467436e-05, "loss": 0.1968, "step": 1200 }, { "epoch": 0.9, "grad_norm": 12.627870559692383, "learning_rate": 3.8939269029953284e-05, "loss": 0.1783, "step": 1210 }, { "epoch": 0.91, "grad_norm": 9.900853157043457, "learning_rate": 3.880186864523221e-05, "loss": 0.2126, "step": 1220 }, { "epoch": 0.91, "grad_norm": 3.763246536254883, "learning_rate": 3.866446826051113e-05, "loss": 0.2136, "step": 1230 }, { "epoch": 0.92, "grad_norm": 7.94022274017334, "learning_rate": 3.852706787579005e-05, "loss": 0.1912, "step": 1240 }, { "epoch": 0.93, "grad_norm": 9.354190826416016, "learning_rate": 3.8389667491068977e-05, "loss": 0.1925, "step": 1250 }, { "epoch": 0.93, "grad_norm": 12.671570777893066, "learning_rate": 3.82522671063479e-05, "loss": 0.1832, "step": 1260 }, { "epoch": 0.94, "grad_norm": 18.882457733154297, "learning_rate": 3.811486672162682e-05, "loss": 0.1868, "step": 1270 }, { "epoch": 0.95, "grad_norm": 8.26636028289795, "learning_rate": 3.797746633690574e-05, "loss": 0.1815, "step": 1280 }, { "epoch": 0.96, "grad_norm": 5.035627841949463, "learning_rate": 3.784006595218467e-05, "loss": 0.1912, "step": 1290 }, { "epoch": 0.96, "grad_norm": 8.453719139099121, "learning_rate": 3.770266556746359e-05, "loss": 0.1617, "step": 1300 }, { "epoch": 0.97, "grad_norm": 6.647568225860596, "learning_rate": 3.7565265182742515e-05, "loss": 0.1969, "step": 1310 }, { "epoch": 0.98, "grad_norm": 7.901888370513916, "learning_rate": 3.7427864798021435e-05, "loss": 0.1893, "step": 1320 }, { "epoch": 0.99, "grad_norm": 5.35443115234375, "learning_rate": 3.7290464413300355e-05, "loss": 0.1674, "step": 1330 }, { "epoch": 0.99, "grad_norm": 8.808985710144043, "learning_rate": 3.715306402857928e-05, "loss": 0.1762, "step": 1340 }, { "epoch": 1.0, "eval_accuracy": 0.947412963585564, "eval_loss": 0.1263597458600998, "eval_runtime": 141.4931, "eval_samples_per_second": 609.62, "eval_steps_per_second": 38.108, "step": 1348 }, { "epoch": 1.0, "grad_norm": 7.0969462394714355, "learning_rate": 3.701566364385821e-05, "loss": 0.1727, "step": 1350 }, { "epoch": 1.01, "grad_norm": 11.357718467712402, "learning_rate": 3.687826325913713e-05, "loss": 0.1529, "step": 1360 }, { "epoch": 1.02, "grad_norm": 16.987056732177734, "learning_rate": 3.674086287441605e-05, "loss": 0.192, "step": 1370 }, { "epoch": 1.02, "grad_norm": 9.532151222229004, "learning_rate": 3.6603462489694974e-05, "loss": 0.1439, "step": 1380 }, { "epoch": 1.03, "grad_norm": 9.81009578704834, "learning_rate": 3.64660621049739e-05, "loss": 0.1541, "step": 1390 }, { "epoch": 1.04, "grad_norm": 6.317430019378662, "learning_rate": 3.632866172025282e-05, "loss": 0.1602, "step": 1400 }, { "epoch": 1.05, "grad_norm": 4.5247039794921875, "learning_rate": 3.619126133553174e-05, "loss": 0.1465, "step": 1410 }, { "epoch": 1.05, "grad_norm": 6.828434944152832, "learning_rate": 3.6053860950810666e-05, "loss": 0.2143, "step": 1420 }, { "epoch": 1.06, "grad_norm": 9.505870819091797, "learning_rate": 3.5916460566089586e-05, "loss": 0.2331, "step": 1430 }, { "epoch": 1.07, "grad_norm": 7.402731418609619, "learning_rate": 3.577906018136851e-05, "loss": 0.1997, "step": 1440 }, { "epoch": 1.08, "grad_norm": 13.928895950317383, "learning_rate": 3.564165979664743e-05, "loss": 0.1778, "step": 1450 }, { "epoch": 1.08, "grad_norm": 7.864881992340088, "learning_rate": 3.550425941192635e-05, "loss": 0.1937, "step": 1460 }, { "epoch": 1.09, "grad_norm": 8.725811004638672, "learning_rate": 3.536685902720528e-05, "loss": 0.166, "step": 1470 }, { "epoch": 1.1, "grad_norm": 19.298831939697266, "learning_rate": 3.5229458642484205e-05, "loss": 0.2126, "step": 1480 }, { "epoch": 1.11, "grad_norm": 6.058204650878906, "learning_rate": 3.5092058257763125e-05, "loss": 0.1524, "step": 1490 }, { "epoch": 1.11, "grad_norm": 8.551276206970215, "learning_rate": 3.4954657873042044e-05, "loss": 0.1853, "step": 1500 }, { "epoch": 1.12, "grad_norm": 11.204520225524902, "learning_rate": 3.481725748832097e-05, "loss": 0.1951, "step": 1510 }, { "epoch": 1.13, "grad_norm": 2.948188543319702, "learning_rate": 3.46798571035999e-05, "loss": 0.1559, "step": 1520 }, { "epoch": 1.14, "grad_norm": 12.01566219329834, "learning_rate": 3.454245671887881e-05, "loss": 0.1783, "step": 1530 }, { "epoch": 1.14, "grad_norm": 15.104211807250977, "learning_rate": 3.440505633415774e-05, "loss": 0.1621, "step": 1540 }, { "epoch": 1.15, "grad_norm": 5.36481237411499, "learning_rate": 3.426765594943666e-05, "loss": 0.172, "step": 1550 }, { "epoch": 1.16, "grad_norm": 5.6568827629089355, "learning_rate": 3.413025556471558e-05, "loss": 0.1645, "step": 1560 }, { "epoch": 1.16, "grad_norm": 11.129305839538574, "learning_rate": 3.39928551799945e-05, "loss": 0.1862, "step": 1570 }, { "epoch": 1.17, "grad_norm": 6.849048137664795, "learning_rate": 3.385545479527343e-05, "loss": 0.1507, "step": 1580 }, { "epoch": 1.18, "grad_norm": 8.361126899719238, "learning_rate": 3.371805441055235e-05, "loss": 0.1668, "step": 1590 }, { "epoch": 1.19, "grad_norm": 3.8484487533569336, "learning_rate": 3.3580654025831275e-05, "loss": 0.1965, "step": 1600 }, { "epoch": 1.19, "grad_norm": 11.742258071899414, "learning_rate": 3.3443253641110195e-05, "loss": 0.1384, "step": 1610 }, { "epoch": 1.2, "grad_norm": 7.576124668121338, "learning_rate": 3.330585325638912e-05, "loss": 0.1316, "step": 1620 }, { "epoch": 1.21, "grad_norm": 15.978646278381348, "learning_rate": 3.316845287166804e-05, "loss": 0.1324, "step": 1630 }, { "epoch": 1.22, "grad_norm": 2.425368070602417, "learning_rate": 3.303105248694697e-05, "loss": 0.1429, "step": 1640 }, { "epoch": 1.22, "grad_norm": 8.935554504394531, "learning_rate": 3.289365210222589e-05, "loss": 0.1087, "step": 1650 }, { "epoch": 1.23, "grad_norm": 6.281069755554199, "learning_rate": 3.275625171750481e-05, "loss": 0.1681, "step": 1660 }, { "epoch": 1.24, "grad_norm": 8.907856941223145, "learning_rate": 3.2618851332783734e-05, "loss": 0.1431, "step": 1670 }, { "epoch": 1.25, "grad_norm": 5.795341968536377, "learning_rate": 3.248145094806266e-05, "loss": 0.1575, "step": 1680 }, { "epoch": 1.25, "grad_norm": 12.340773582458496, "learning_rate": 3.234405056334158e-05, "loss": 0.1537, "step": 1690 }, { "epoch": 1.26, "grad_norm": 9.601103782653809, "learning_rate": 3.22066501786205e-05, "loss": 0.1618, "step": 1700 }, { "epoch": 1.27, "grad_norm": 9.238292694091797, "learning_rate": 3.2069249793899426e-05, "loss": 0.15, "step": 1710 }, { "epoch": 1.28, "grad_norm": 6.925644874572754, "learning_rate": 3.193184940917835e-05, "loss": 0.1913, "step": 1720 }, { "epoch": 1.28, "grad_norm": 4.840887069702148, "learning_rate": 3.1794449024457266e-05, "loss": 0.1547, "step": 1730 }, { "epoch": 1.29, "grad_norm": 11.813157081604004, "learning_rate": 3.165704863973619e-05, "loss": 0.1488, "step": 1740 }, { "epoch": 1.3, "grad_norm": 16.899797439575195, "learning_rate": 3.151964825501512e-05, "loss": 0.1665, "step": 1750 }, { "epoch": 1.31, "grad_norm": 10.905778884887695, "learning_rate": 3.138224787029404e-05, "loss": 0.1484, "step": 1760 }, { "epoch": 1.31, "grad_norm": 14.453968048095703, "learning_rate": 3.124484748557296e-05, "loss": 0.1327, "step": 1770 }, { "epoch": 1.32, "grad_norm": 9.780988693237305, "learning_rate": 3.1107447100851885e-05, "loss": 0.103, "step": 1780 }, { "epoch": 1.33, "grad_norm": 4.881911277770996, "learning_rate": 3.0970046716130804e-05, "loss": 0.1604, "step": 1790 }, { "epoch": 1.34, "grad_norm": 4.467998027801514, "learning_rate": 3.083264633140973e-05, "loss": 0.1167, "step": 1800 }, { "epoch": 1.34, "grad_norm": 11.12286376953125, "learning_rate": 3.069524594668865e-05, "loss": 0.139, "step": 1810 }, { "epoch": 1.35, "grad_norm": 8.770483016967773, "learning_rate": 3.055784556196758e-05, "loss": 0.13, "step": 1820 }, { "epoch": 1.36, "grad_norm": 9.070440292358398, "learning_rate": 3.0420445177246497e-05, "loss": 0.1492, "step": 1830 }, { "epoch": 1.36, "grad_norm": 8.814356803894043, "learning_rate": 3.0283044792525423e-05, "loss": 0.1872, "step": 1840 }, { "epoch": 1.37, "grad_norm": 7.875304222106934, "learning_rate": 3.014564440780434e-05, "loss": 0.1561, "step": 1850 }, { "epoch": 1.38, "grad_norm": 6.7557692527771, "learning_rate": 3.0008244023083266e-05, "loss": 0.1345, "step": 1860 }, { "epoch": 1.39, "grad_norm": 11.669951438903809, "learning_rate": 2.987084363836219e-05, "loss": 0.1003, "step": 1870 }, { "epoch": 1.39, "grad_norm": 9.418144226074219, "learning_rate": 2.9733443253641112e-05, "loss": 0.1245, "step": 1880 }, { "epoch": 1.4, "grad_norm": 5.591629505157471, "learning_rate": 2.9596042868920032e-05, "loss": 0.1536, "step": 1890 }, { "epoch": 1.41, "grad_norm": 10.448158264160156, "learning_rate": 2.9458642484198955e-05, "loss": 0.1217, "step": 1900 }, { "epoch": 1.42, "grad_norm": 8.35586166381836, "learning_rate": 2.932124209947788e-05, "loss": 0.1517, "step": 1910 }, { "epoch": 1.42, "grad_norm": 10.111129760742188, "learning_rate": 2.9183841714756805e-05, "loss": 0.1271, "step": 1920 }, { "epoch": 1.43, "grad_norm": 3.2071399688720703, "learning_rate": 2.9046441330035725e-05, "loss": 0.118, "step": 1930 }, { "epoch": 1.44, "grad_norm": 22.524978637695312, "learning_rate": 2.8909040945314648e-05, "loss": 0.1511, "step": 1940 }, { "epoch": 1.45, "grad_norm": 10.672737121582031, "learning_rate": 2.877164056059357e-05, "loss": 0.1224, "step": 1950 }, { "epoch": 1.45, "grad_norm": 10.548683166503906, "learning_rate": 2.8634240175872494e-05, "loss": 0.1307, "step": 1960 }, { "epoch": 1.46, "grad_norm": 8.707009315490723, "learning_rate": 2.8496839791151414e-05, "loss": 0.1184, "step": 1970 }, { "epoch": 1.47, "grad_norm": 4.6339874267578125, "learning_rate": 2.8359439406430337e-05, "loss": 0.143, "step": 1980 }, { "epoch": 1.48, "grad_norm": 11.373931884765625, "learning_rate": 2.8222039021709263e-05, "loss": 0.1468, "step": 1990 }, { "epoch": 1.48, "grad_norm": 5.647881031036377, "learning_rate": 2.8084638636988186e-05, "loss": 0.1188, "step": 2000 }, { "epoch": 1.49, "grad_norm": 9.399796485900879, "learning_rate": 2.7947238252267106e-05, "loss": 0.1484, "step": 2010 }, { "epoch": 1.5, "grad_norm": 11.064074516296387, "learning_rate": 2.780983786754603e-05, "loss": 0.1471, "step": 2020 }, { "epoch": 1.51, "grad_norm": 5.499329090118408, "learning_rate": 2.7672437482824952e-05, "loss": 0.136, "step": 2030 }, { "epoch": 1.51, "grad_norm": 12.6922025680542, "learning_rate": 2.753503709810388e-05, "loss": 0.1292, "step": 2040 }, { "epoch": 1.52, "grad_norm": 6.548159599304199, "learning_rate": 2.7397636713382795e-05, "loss": 0.0878, "step": 2050 }, { "epoch": 1.53, "grad_norm": 11.498106002807617, "learning_rate": 2.726023632866172e-05, "loss": 0.1235, "step": 2060 }, { "epoch": 1.54, "grad_norm": 14.733744621276855, "learning_rate": 2.7122835943940645e-05, "loss": 0.1243, "step": 2070 }, { "epoch": 1.54, "grad_norm": 4.867096424102783, "learning_rate": 2.6985435559219568e-05, "loss": 0.1553, "step": 2080 }, { "epoch": 1.55, "grad_norm": 18.455617904663086, "learning_rate": 2.684803517449849e-05, "loss": 0.161, "step": 2090 }, { "epoch": 1.56, "grad_norm": 7.310939311981201, "learning_rate": 2.671063478977741e-05, "loss": 0.1245, "step": 2100 }, { "epoch": 1.57, "grad_norm": 6.81366491317749, "learning_rate": 2.6573234405056334e-05, "loss": 0.1673, "step": 2110 }, { "epoch": 1.57, "grad_norm": 10.377093315124512, "learning_rate": 2.643583402033526e-05, "loss": 0.1255, "step": 2120 }, { "epoch": 1.58, "grad_norm": 8.426204681396484, "learning_rate": 2.6298433635614183e-05, "loss": 0.158, "step": 2130 }, { "epoch": 1.59, "grad_norm": 4.7572760581970215, "learning_rate": 2.6161033250893103e-05, "loss": 0.1573, "step": 2140 }, { "epoch": 1.59, "grad_norm": 12.727173805236816, "learning_rate": 2.6023632866172026e-05, "loss": 0.1426, "step": 2150 }, { "epoch": 1.6, "grad_norm": 8.446662902832031, "learning_rate": 2.588623248145095e-05, "loss": 0.1689, "step": 2160 }, { "epoch": 1.61, "grad_norm": 5.482067584991455, "learning_rate": 2.5748832096729876e-05, "loss": 0.1257, "step": 2170 }, { "epoch": 1.62, "grad_norm": 12.713666915893555, "learning_rate": 2.5611431712008792e-05, "loss": 0.1682, "step": 2180 }, { "epoch": 1.62, "grad_norm": 6.469407081604004, "learning_rate": 2.547403132728772e-05, "loss": 0.1454, "step": 2190 }, { "epoch": 1.63, "grad_norm": 14.35875129699707, "learning_rate": 2.5336630942566642e-05, "loss": 0.1278, "step": 2200 }, { "epoch": 1.64, "grad_norm": 9.981734275817871, "learning_rate": 2.5199230557845565e-05, "loss": 0.1292, "step": 2210 }, { "epoch": 1.65, "grad_norm": 5.275669574737549, "learning_rate": 2.5061830173124485e-05, "loss": 0.1384, "step": 2220 }, { "epoch": 1.65, "grad_norm": 8.421156883239746, "learning_rate": 2.4924429788403408e-05, "loss": 0.0778, "step": 2230 }, { "epoch": 1.66, "grad_norm": 6.505790710449219, "learning_rate": 2.478702940368233e-05, "loss": 0.1316, "step": 2240 }, { "epoch": 1.67, "grad_norm": 13.836162567138672, "learning_rate": 2.4649629018961254e-05, "loss": 0.1066, "step": 2250 }, { "epoch": 1.68, "grad_norm": 13.854897499084473, "learning_rate": 2.4512228634240177e-05, "loss": 0.1359, "step": 2260 }, { "epoch": 1.68, "grad_norm": 8.590911865234375, "learning_rate": 2.43748282495191e-05, "loss": 0.1295, "step": 2270 }, { "epoch": 1.69, "grad_norm": 3.8553526401519775, "learning_rate": 2.4237427864798023e-05, "loss": 0.1308, "step": 2280 }, { "epoch": 1.7, "grad_norm": 5.948857307434082, "learning_rate": 2.4100027480076946e-05, "loss": 0.1335, "step": 2290 }, { "epoch": 1.71, "grad_norm": 10.528166770935059, "learning_rate": 2.396262709535587e-05, "loss": 0.1331, "step": 2300 }, { "epoch": 1.71, "grad_norm": 3.8531999588012695, "learning_rate": 2.382522671063479e-05, "loss": 0.1208, "step": 2310 }, { "epoch": 1.72, "grad_norm": 15.348362922668457, "learning_rate": 2.3687826325913716e-05, "loss": 0.1525, "step": 2320 }, { "epoch": 1.73, "grad_norm": 9.51009464263916, "learning_rate": 2.3550425941192635e-05, "loss": 0.118, "step": 2330 }, { "epoch": 1.74, "grad_norm": 13.064322471618652, "learning_rate": 2.341302555647156e-05, "loss": 0.1434, "step": 2340 }, { "epoch": 1.74, "grad_norm": 5.5084614753723145, "learning_rate": 2.327562517175048e-05, "loss": 0.1387, "step": 2350 }, { "epoch": 1.75, "grad_norm": 10.431999206542969, "learning_rate": 2.3138224787029405e-05, "loss": 0.1147, "step": 2360 }, { "epoch": 1.76, "grad_norm": 14.113656044006348, "learning_rate": 2.3000824402308328e-05, "loss": 0.1442, "step": 2370 }, { "epoch": 1.77, "grad_norm": 8.030630111694336, "learning_rate": 2.286342401758725e-05, "loss": 0.1294, "step": 2380 }, { "epoch": 1.77, "grad_norm": 5.992265224456787, "learning_rate": 2.272602363286617e-05, "loss": 0.0894, "step": 2390 }, { "epoch": 1.78, "grad_norm": 10.719809532165527, "learning_rate": 2.2588623248145097e-05, "loss": 0.107, "step": 2400 }, { "epoch": 1.79, "grad_norm": 13.229023933410645, "learning_rate": 2.2451222863424017e-05, "loss": 0.1236, "step": 2410 }, { "epoch": 1.8, "grad_norm": 9.23493766784668, "learning_rate": 2.2313822478702943e-05, "loss": 0.1576, "step": 2420 }, { "epoch": 1.8, "grad_norm": 12.941469192504883, "learning_rate": 2.2176422093981863e-05, "loss": 0.1393, "step": 2430 }, { "epoch": 1.81, "grad_norm": 7.724160194396973, "learning_rate": 2.2039021709260786e-05, "loss": 0.1096, "step": 2440 }, { "epoch": 1.82, "grad_norm": 5.428783416748047, "learning_rate": 2.190162132453971e-05, "loss": 0.1035, "step": 2450 }, { "epoch": 1.82, "grad_norm": 14.180803298950195, "learning_rate": 2.1764220939818633e-05, "loss": 0.1444, "step": 2460 }, { "epoch": 1.83, "grad_norm": 7.265615940093994, "learning_rate": 2.1626820555097556e-05, "loss": 0.1496, "step": 2470 }, { "epoch": 1.84, "grad_norm": 9.01385498046875, "learning_rate": 2.148942017037648e-05, "loss": 0.1124, "step": 2480 }, { "epoch": 1.85, "grad_norm": 7.296875476837158, "learning_rate": 2.13520197856554e-05, "loss": 0.1387, "step": 2490 }, { "epoch": 1.85, "grad_norm": 12.550304412841797, "learning_rate": 2.1214619400934325e-05, "loss": 0.105, "step": 2500 }, { "epoch": 1.86, "grad_norm": 5.9940385818481445, "learning_rate": 2.1077219016213245e-05, "loss": 0.1368, "step": 2510 }, { "epoch": 1.87, "grad_norm": 5.355954170227051, "learning_rate": 2.093981863149217e-05, "loss": 0.086, "step": 2520 }, { "epoch": 1.88, "grad_norm": 8.216740608215332, "learning_rate": 2.080241824677109e-05, "loss": 0.0973, "step": 2530 }, { "epoch": 1.88, "grad_norm": 9.250962257385254, "learning_rate": 2.0665017862050014e-05, "loss": 0.103, "step": 2540 }, { "epoch": 1.89, "grad_norm": 12.288225173950195, "learning_rate": 2.0527617477328937e-05, "loss": 0.1405, "step": 2550 }, { "epoch": 1.9, "grad_norm": 8.24113941192627, "learning_rate": 2.039021709260786e-05, "loss": 0.1371, "step": 2560 }, { "epoch": 1.91, "grad_norm": 24.87588882446289, "learning_rate": 2.0252816707886783e-05, "loss": 0.1137, "step": 2570 }, { "epoch": 1.91, "grad_norm": 13.435257911682129, "learning_rate": 2.0115416323165706e-05, "loss": 0.1034, "step": 2580 }, { "epoch": 1.92, "grad_norm": 18.450393676757812, "learning_rate": 1.9978015938444626e-05, "loss": 0.1475, "step": 2590 }, { "epoch": 1.93, "grad_norm": 8.536957740783691, "learning_rate": 1.9840615553723553e-05, "loss": 0.0997, "step": 2600 }, { "epoch": 1.94, "grad_norm": 9.833585739135742, "learning_rate": 1.9703215169002472e-05, "loss": 0.1198, "step": 2610 }, { "epoch": 1.94, "grad_norm": 6.295921802520752, "learning_rate": 1.9565814784281396e-05, "loss": 0.1138, "step": 2620 }, { "epoch": 1.95, "grad_norm": 9.540913581848145, "learning_rate": 1.942841439956032e-05, "loss": 0.0946, "step": 2630 }, { "epoch": 1.96, "grad_norm": 5.700273036956787, "learning_rate": 1.9291014014839242e-05, "loss": 0.1178, "step": 2640 }, { "epoch": 1.97, "grad_norm": 8.3849515914917, "learning_rate": 1.9153613630118165e-05, "loss": 0.1156, "step": 2650 }, { "epoch": 1.97, "grad_norm": 15.1934232711792, "learning_rate": 1.9016213245397088e-05, "loss": 0.1263, "step": 2660 }, { "epoch": 1.98, "grad_norm": 8.725594520568848, "learning_rate": 1.887881286067601e-05, "loss": 0.1151, "step": 2670 }, { "epoch": 1.99, "grad_norm": 9.016326904296875, "learning_rate": 1.8741412475954934e-05, "loss": 0.1424, "step": 2680 }, { "epoch": 2.0, "grad_norm": 5.650193214416504, "learning_rate": 1.8604012091233854e-05, "loss": 0.1407, "step": 2690 }, { "epoch": 2.0, "eval_accuracy": 0.972523969069177, "eval_loss": 0.07045618444681168, "eval_runtime": 140.2638, "eval_samples_per_second": 614.963, "eval_steps_per_second": 38.442, "step": 2696 }, { "epoch": 2.0, "grad_norm": 4.413318634033203, "learning_rate": 1.846661170651278e-05, "loss": 0.1072, "step": 2700 }, { "epoch": 2.01, "grad_norm": 6.745368003845215, "learning_rate": 1.83292113217917e-05, "loss": 0.1047, "step": 2710 }, { "epoch": 2.02, "grad_norm": 7.095730304718018, "learning_rate": 1.8191810937070623e-05, "loss": 0.1024, "step": 2720 }, { "epoch": 2.03, "grad_norm": 13.053765296936035, "learning_rate": 1.805441055234955e-05, "loss": 0.1113, "step": 2730 }, { "epoch": 2.03, "grad_norm": 10.390790939331055, "learning_rate": 1.791701016762847e-05, "loss": 0.1106, "step": 2740 }, { "epoch": 2.04, "grad_norm": 8.856596946716309, "learning_rate": 1.7779609782907393e-05, "loss": 0.0861, "step": 2750 }, { "epoch": 2.05, "grad_norm": 4.328437805175781, "learning_rate": 1.7642209398186316e-05, "loss": 0.0798, "step": 2760 }, { "epoch": 2.05, "grad_norm": 4.761692523956299, "learning_rate": 1.750480901346524e-05, "loss": 0.0879, "step": 2770 }, { "epoch": 2.06, "grad_norm": 12.073725700378418, "learning_rate": 1.7367408628744162e-05, "loss": 0.0962, "step": 2780 }, { "epoch": 2.07, "grad_norm": 1.6801289319992065, "learning_rate": 1.7230008244023085e-05, "loss": 0.0877, "step": 2790 }, { "epoch": 2.08, "grad_norm": 12.418718338012695, "learning_rate": 1.7092607859302008e-05, "loss": 0.0984, "step": 2800 }, { "epoch": 2.08, "grad_norm": 3.915785312652588, "learning_rate": 1.695520747458093e-05, "loss": 0.108, "step": 2810 }, { "epoch": 2.09, "grad_norm": 6.9176435470581055, "learning_rate": 1.681780708985985e-05, "loss": 0.1026, "step": 2820 }, { "epoch": 2.1, "grad_norm": 3.9952585697174072, "learning_rate": 1.6680406705138778e-05, "loss": 0.0622, "step": 2830 }, { "epoch": 2.11, "grad_norm": 11.399799346923828, "learning_rate": 1.6543006320417697e-05, "loss": 0.1219, "step": 2840 }, { "epoch": 2.11, "grad_norm": 9.78951358795166, "learning_rate": 1.640560593569662e-05, "loss": 0.1127, "step": 2850 }, { "epoch": 2.12, "grad_norm": 5.477469444274902, "learning_rate": 1.6268205550975543e-05, "loss": 0.0797, "step": 2860 }, { "epoch": 2.13, "grad_norm": 8.457229614257812, "learning_rate": 1.6130805166254467e-05, "loss": 0.1041, "step": 2870 }, { "epoch": 2.14, "grad_norm": 11.861204147338867, "learning_rate": 1.599340478153339e-05, "loss": 0.1057, "step": 2880 }, { "epoch": 2.14, "grad_norm": 10.859254837036133, "learning_rate": 1.5856004396812313e-05, "loss": 0.0783, "step": 2890 }, { "epoch": 2.15, "grad_norm": 7.593705177307129, "learning_rate": 1.5718604012091236e-05, "loss": 0.101, "step": 2900 }, { "epoch": 2.16, "grad_norm": 11.803435325622559, "learning_rate": 1.558120362737016e-05, "loss": 0.0835, "step": 2910 }, { "epoch": 2.17, "grad_norm": 10.460546493530273, "learning_rate": 1.544380324264908e-05, "loss": 0.0951, "step": 2920 }, { "epoch": 2.17, "grad_norm": 14.229741096496582, "learning_rate": 1.5306402857928005e-05, "loss": 0.1109, "step": 2930 }, { "epoch": 2.18, "grad_norm": 4.7770233154296875, "learning_rate": 1.5169002473206925e-05, "loss": 0.1304, "step": 2940 }, { "epoch": 2.19, "grad_norm": 9.118112564086914, "learning_rate": 1.503160208848585e-05, "loss": 0.0583, "step": 2950 }, { "epoch": 2.2, "grad_norm": 5.745917797088623, "learning_rate": 1.4894201703764771e-05, "loss": 0.0851, "step": 2960 }, { "epoch": 2.2, "grad_norm": 8.237546920776367, "learning_rate": 1.4756801319043694e-05, "loss": 0.106, "step": 2970 }, { "epoch": 2.21, "grad_norm": 12.706246376037598, "learning_rate": 1.4619400934322616e-05, "loss": 0.1187, "step": 2980 }, { "epoch": 2.22, "grad_norm": 6.276710033416748, "learning_rate": 1.448200054960154e-05, "loss": 0.1031, "step": 2990 }, { "epoch": 2.23, "grad_norm": 4.847949981689453, "learning_rate": 1.4344600164880462e-05, "loss": 0.0694, "step": 3000 }, { "epoch": 2.23, "grad_norm": 5.567623615264893, "learning_rate": 1.4207199780159385e-05, "loss": 0.0809, "step": 3010 }, { "epoch": 2.24, "grad_norm": 5.528212070465088, "learning_rate": 1.4069799395438306e-05, "loss": 0.1021, "step": 3020 }, { "epoch": 2.25, "grad_norm": 10.893386840820312, "learning_rate": 1.3932399010717231e-05, "loss": 0.0918, "step": 3030 }, { "epoch": 2.26, "grad_norm": 13.184591293334961, "learning_rate": 1.3794998625996153e-05, "loss": 0.1005, "step": 3040 }, { "epoch": 2.26, "grad_norm": 7.119046688079834, "learning_rate": 1.3657598241275078e-05, "loss": 0.0996, "step": 3050 }, { "epoch": 2.27, "grad_norm": 12.941669464111328, "learning_rate": 1.3520197856553999e-05, "loss": 0.084, "step": 3060 }, { "epoch": 2.28, "grad_norm": 8.232301712036133, "learning_rate": 1.3382797471832922e-05, "loss": 0.0887, "step": 3070 }, { "epoch": 2.28, "grad_norm": 9.092889785766602, "learning_rate": 1.3245397087111843e-05, "loss": 0.0794, "step": 3080 }, { "epoch": 2.29, "grad_norm": 7.75855827331543, "learning_rate": 1.3107996702390768e-05, "loss": 0.0795, "step": 3090 }, { "epoch": 2.3, "grad_norm": 14.936914443969727, "learning_rate": 1.297059631766969e-05, "loss": 0.0687, "step": 3100 }, { "epoch": 2.31, "grad_norm": 9.405156135559082, "learning_rate": 1.2833195932948613e-05, "loss": 0.113, "step": 3110 }, { "epoch": 2.31, "grad_norm": 10.729270935058594, "learning_rate": 1.2695795548227534e-05, "loss": 0.1129, "step": 3120 }, { "epoch": 2.32, "grad_norm": 11.434041976928711, "learning_rate": 1.2558395163506459e-05, "loss": 0.0989, "step": 3130 }, { "epoch": 2.33, "grad_norm": 8.561604499816895, "learning_rate": 1.2420994778785382e-05, "loss": 0.0948, "step": 3140 }, { "epoch": 2.34, "grad_norm": 6.7873406410217285, "learning_rate": 1.2283594394064305e-05, "loss": 0.0843, "step": 3150 }, { "epoch": 2.34, "grad_norm": 13.5293607711792, "learning_rate": 1.2146194009343227e-05, "loss": 0.1164, "step": 3160 }, { "epoch": 2.35, "grad_norm": 9.595284461975098, "learning_rate": 1.200879362462215e-05, "loss": 0.0833, "step": 3170 }, { "epoch": 2.36, "grad_norm": 7.143069267272949, "learning_rate": 1.1871393239901073e-05, "loss": 0.1036, "step": 3180 }, { "epoch": 2.37, "grad_norm": 12.40601921081543, "learning_rate": 1.1733992855179996e-05, "loss": 0.0641, "step": 3190 }, { "epoch": 2.37, "grad_norm": 5.299369812011719, "learning_rate": 1.1596592470458917e-05, "loss": 0.0835, "step": 3200 }, { "epoch": 2.38, "grad_norm": 3.2412006855010986, "learning_rate": 1.145919208573784e-05, "loss": 0.0985, "step": 3210 }, { "epoch": 2.39, "grad_norm": 9.663202285766602, "learning_rate": 1.1321791701016764e-05, "loss": 0.0725, "step": 3220 }, { "epoch": 2.4, "grad_norm": 9.806108474731445, "learning_rate": 1.1184391316295687e-05, "loss": 0.0976, "step": 3230 }, { "epoch": 2.4, "grad_norm": 12.536150932312012, "learning_rate": 1.104699093157461e-05, "loss": 0.1035, "step": 3240 }, { "epoch": 2.41, "grad_norm": 19.915842056274414, "learning_rate": 1.0909590546853531e-05, "loss": 0.0878, "step": 3250 }, { "epoch": 2.42, "grad_norm": 3.88319993019104, "learning_rate": 1.0772190162132454e-05, "loss": 0.0788, "step": 3260 }, { "epoch": 2.43, "grad_norm": 6.242338180541992, "learning_rate": 1.0634789777411378e-05, "loss": 0.096, "step": 3270 }, { "epoch": 2.43, "grad_norm": 10.742528915405273, "learning_rate": 1.04973893926903e-05, "loss": 0.0954, "step": 3280 }, { "epoch": 2.44, "grad_norm": 4.696013927459717, "learning_rate": 1.0359989007969224e-05, "loss": 0.0945, "step": 3290 }, { "epoch": 2.45, "grad_norm": 4.943475723266602, "learning_rate": 1.0222588623248145e-05, "loss": 0.1001, "step": 3300 }, { "epoch": 2.46, "grad_norm": 10.133197784423828, "learning_rate": 1.0085188238527068e-05, "loss": 0.0921, "step": 3310 }, { "epoch": 2.46, "grad_norm": 11.798482894897461, "learning_rate": 9.947787853805991e-06, "loss": 0.0675, "step": 3320 }, { "epoch": 2.47, "grad_norm": 5.647496700286865, "learning_rate": 9.810387469084915e-06, "loss": 0.0721, "step": 3330 }, { "epoch": 2.48, "grad_norm": 3.0277674198150635, "learning_rate": 9.672987084363836e-06, "loss": 0.0798, "step": 3340 }, { "epoch": 2.49, "grad_norm": 5.510377407073975, "learning_rate": 9.535586699642759e-06, "loss": 0.1071, "step": 3350 }, { "epoch": 2.49, "grad_norm": 7.622246265411377, "learning_rate": 9.398186314921682e-06, "loss": 0.0813, "step": 3360 }, { "epoch": 2.5, "grad_norm": 12.018797874450684, "learning_rate": 9.260785930200605e-06, "loss": 0.0849, "step": 3370 }, { "epoch": 2.51, "grad_norm": 5.821876525878906, "learning_rate": 9.123385545479528e-06, "loss": 0.0701, "step": 3380 }, { "epoch": 2.51, "grad_norm": 12.286214828491211, "learning_rate": 8.98598516075845e-06, "loss": 0.0825, "step": 3390 }, { "epoch": 2.52, "grad_norm": 5.903075218200684, "learning_rate": 8.848584776037373e-06, "loss": 0.0981, "step": 3400 }, { "epoch": 2.53, "grad_norm": 5.024692058563232, "learning_rate": 8.711184391316296e-06, "loss": 0.0905, "step": 3410 }, { "epoch": 2.54, "grad_norm": 14.582296371459961, "learning_rate": 8.573784006595219e-06, "loss": 0.1047, "step": 3420 }, { "epoch": 2.54, "grad_norm": 4.112821578979492, "learning_rate": 8.436383621874142e-06, "loss": 0.0831, "step": 3430 }, { "epoch": 2.55, "grad_norm": 8.43770980834961, "learning_rate": 8.298983237153064e-06, "loss": 0.0961, "step": 3440 }, { "epoch": 2.56, "grad_norm": 10.004687309265137, "learning_rate": 8.161582852431987e-06, "loss": 0.08, "step": 3450 }, { "epoch": 2.57, "grad_norm": 4.888688564300537, "learning_rate": 8.02418246771091e-06, "loss": 0.0741, "step": 3460 }, { "epoch": 2.57, "grad_norm": 6.9248881340026855, "learning_rate": 7.886782082989833e-06, "loss": 0.0918, "step": 3470 }, { "epoch": 2.58, "grad_norm": 5.887302398681641, "learning_rate": 7.749381698268756e-06, "loss": 0.0693, "step": 3480 }, { "epoch": 2.59, "grad_norm": 8.469816207885742, "learning_rate": 7.611981313547678e-06, "loss": 0.0911, "step": 3490 }, { "epoch": 2.6, "grad_norm": 1.2210373878479004, "learning_rate": 7.474580928826601e-06, "loss": 0.0667, "step": 3500 }, { "epoch": 2.6, "grad_norm": 10.158321380615234, "learning_rate": 7.337180544105524e-06, "loss": 0.0658, "step": 3510 }, { "epoch": 2.61, "grad_norm": 8.220787048339844, "learning_rate": 7.199780159384446e-06, "loss": 0.088, "step": 3520 }, { "epoch": 2.62, "grad_norm": 10.971105575561523, "learning_rate": 7.062379774663369e-06, "loss": 0.1002, "step": 3530 }, { "epoch": 2.63, "grad_norm": 8.404732704162598, "learning_rate": 6.924979389942292e-06, "loss": 0.0869, "step": 3540 }, { "epoch": 2.63, "grad_norm": 8.824265480041504, "learning_rate": 6.7875790052212145e-06, "loss": 0.0705, "step": 3550 }, { "epoch": 2.64, "grad_norm": 3.96179461479187, "learning_rate": 6.650178620500138e-06, "loss": 0.0603, "step": 3560 }, { "epoch": 2.65, "grad_norm": 8.119577407836914, "learning_rate": 6.51277823577906e-06, "loss": 0.0719, "step": 3570 }, { "epoch": 2.66, "grad_norm": 7.250158786773682, "learning_rate": 6.375377851057983e-06, "loss": 0.0906, "step": 3580 }, { "epoch": 2.66, "grad_norm": 4.2896623611450195, "learning_rate": 6.237977466336906e-06, "loss": 0.0977, "step": 3590 }, { "epoch": 2.67, "grad_norm": 2.2607102394104004, "learning_rate": 6.100577081615829e-06, "loss": 0.0627, "step": 3600 }, { "epoch": 2.68, "grad_norm": 4.072081565856934, "learning_rate": 5.9631766968947515e-06, "loss": 0.0909, "step": 3610 }, { "epoch": 2.69, "grad_norm": 7.080986022949219, "learning_rate": 5.825776312173675e-06, "loss": 0.0982, "step": 3620 }, { "epoch": 2.69, "grad_norm": 9.746908187866211, "learning_rate": 5.688375927452598e-06, "loss": 0.0614, "step": 3630 }, { "epoch": 2.7, "grad_norm": 8.818674087524414, "learning_rate": 5.55097554273152e-06, "loss": 0.124, "step": 3640 }, { "epoch": 2.71, "grad_norm": 13.782171249389648, "learning_rate": 5.413575158010443e-06, "loss": 0.0926, "step": 3650 }, { "epoch": 2.72, "grad_norm": 8.21611499786377, "learning_rate": 5.276174773289365e-06, "loss": 0.0693, "step": 3660 }, { "epoch": 2.72, "grad_norm": 10.9508695602417, "learning_rate": 5.1387743885682885e-06, "loss": 0.0974, "step": 3670 }, { "epoch": 2.73, "grad_norm": 14.169780731201172, "learning_rate": 5.001374003847211e-06, "loss": 0.1044, "step": 3680 }, { "epoch": 2.74, "grad_norm": 3.944868564605713, "learning_rate": 4.863973619126134e-06, "loss": 0.0705, "step": 3690 }, { "epoch": 2.74, "grad_norm": 6.739749908447266, "learning_rate": 4.726573234405057e-06, "loss": 0.0679, "step": 3700 }, { "epoch": 2.75, "grad_norm": 13.893156051635742, "learning_rate": 4.589172849683979e-06, "loss": 0.0596, "step": 3710 }, { "epoch": 2.76, "grad_norm": 8.715542793273926, "learning_rate": 4.451772464962902e-06, "loss": 0.0716, "step": 3720 }, { "epoch": 2.77, "grad_norm": 11.366250038146973, "learning_rate": 4.314372080241825e-06, "loss": 0.0949, "step": 3730 }, { "epoch": 2.77, "grad_norm": 5.514875411987305, "learning_rate": 4.176971695520748e-06, "loss": 0.0752, "step": 3740 }, { "epoch": 2.78, "grad_norm": 14.893842697143555, "learning_rate": 4.03957131079967e-06, "loss": 0.108, "step": 3750 }, { "epoch": 2.79, "grad_norm": 12.94680118560791, "learning_rate": 3.902170926078593e-06, "loss": 0.1044, "step": 3760 }, { "epoch": 2.8, "grad_norm": 8.319701194763184, "learning_rate": 3.7647705413575158e-06, "loss": 0.0713, "step": 3770 }, { "epoch": 2.8, "grad_norm": 17.827917098999023, "learning_rate": 3.6273701566364385e-06, "loss": 0.0807, "step": 3780 }, { "epoch": 2.81, "grad_norm": 6.6090312004089355, "learning_rate": 3.489969771915361e-06, "loss": 0.0821, "step": 3790 }, { "epoch": 2.82, "grad_norm": 3.432737112045288, "learning_rate": 3.3525693871942843e-06, "loss": 0.0702, "step": 3800 }, { "epoch": 2.83, "grad_norm": 6.590944766998291, "learning_rate": 3.215169002473207e-06, "loss": 0.0575, "step": 3810 }, { "epoch": 2.83, "grad_norm": 7.214500427246094, "learning_rate": 3.07776861775213e-06, "loss": 0.0532, "step": 3820 }, { "epoch": 2.84, "grad_norm": 1.6727337837219238, "learning_rate": 2.9403682330310528e-06, "loss": 0.0604, "step": 3830 }, { "epoch": 2.85, "grad_norm": 5.265418529510498, "learning_rate": 2.8029678483099755e-06, "loss": 0.0831, "step": 3840 }, { "epoch": 2.86, "grad_norm": 2.7963054180145264, "learning_rate": 2.665567463588898e-06, "loss": 0.0854, "step": 3850 }, { "epoch": 2.86, "grad_norm": 3.112823963165283, "learning_rate": 2.528167078867821e-06, "loss": 0.0575, "step": 3860 }, { "epoch": 2.87, "grad_norm": 7.16569185256958, "learning_rate": 2.390766694146744e-06, "loss": 0.081, "step": 3870 }, { "epoch": 2.88, "grad_norm": 1.7017323970794678, "learning_rate": 2.2533663094256666e-06, "loss": 0.082, "step": 3880 }, { "epoch": 2.89, "grad_norm": 15.373734474182129, "learning_rate": 2.1159659247045893e-06, "loss": 0.0652, "step": 3890 }, { "epoch": 2.89, "grad_norm": 15.30858039855957, "learning_rate": 1.978565539983512e-06, "loss": 0.0755, "step": 3900 }, { "epoch": 2.9, "grad_norm": 10.219656944274902, "learning_rate": 1.8411651552624347e-06, "loss": 0.0774, "step": 3910 }, { "epoch": 2.91, "grad_norm": 6.792389869689941, "learning_rate": 1.7037647705413576e-06, "loss": 0.0773, "step": 3920 }, { "epoch": 2.92, "grad_norm": 13.666367530822754, "learning_rate": 1.5663643858202803e-06, "loss": 0.0614, "step": 3930 }, { "epoch": 2.92, "grad_norm": 7.227494716644287, "learning_rate": 1.4289640010992032e-06, "loss": 0.0831, "step": 3940 }, { "epoch": 2.93, "grad_norm": 7.196842670440674, "learning_rate": 1.2915636163781259e-06, "loss": 0.0511, "step": 3950 }, { "epoch": 2.94, "grad_norm": 6.138864994049072, "learning_rate": 1.1541632316570488e-06, "loss": 0.0722, "step": 3960 }, { "epoch": 2.95, "grad_norm": 7.251946449279785, "learning_rate": 1.0167628469359715e-06, "loss": 0.0812, "step": 3970 }, { "epoch": 2.95, "grad_norm": 9.087138175964355, "learning_rate": 8.793624622148942e-07, "loss": 0.075, "step": 3980 }, { "epoch": 2.96, "grad_norm": 7.0772199630737305, "learning_rate": 7.419620774938171e-07, "loss": 0.0713, "step": 3990 }, { "epoch": 2.97, "grad_norm": 13.134087562561035, "learning_rate": 6.045616927727397e-07, "loss": 0.0738, "step": 4000 }, { "epoch": 2.97, "grad_norm": 4.489412307739258, "learning_rate": 4.6716130805166254e-07, "loss": 0.0742, "step": 4010 }, { "epoch": 2.98, "grad_norm": 3.1018354892730713, "learning_rate": 3.2976092333058533e-07, "loss": 0.0673, "step": 4020 }, { "epoch": 2.99, "grad_norm": 10.936025619506836, "learning_rate": 1.9236053860950813e-07, "loss": 0.0674, "step": 4030 }, { "epoch": 3.0, "grad_norm": 5.309031009674072, "learning_rate": 5.4960153888430885e-08, "loss": 0.0852, "step": 4040 }, { "epoch": 3.0, "eval_accuracy": 0.9823666485038896, "eval_loss": 0.046593740582466125, "eval_runtime": 138.9154, "eval_samples_per_second": 620.932, "eval_steps_per_second": 38.815, "step": 4044 }, { "epoch": 3.0, "step": 4044, "total_flos": 6.432009125858943e+18, "train_loss": 0.17701233503004327, "train_runtime": 1261.8873, "train_samples_per_second": 205.067, "train_steps_per_second": 3.205 } ], "logging_steps": 10, "max_steps": 4044, "num_input_tokens_seen": 0, "num_train_epochs": 3, "save_steps": 500, "total_flos": 6.432009125858943e+18, "train_batch_size": 16, "trial_name": null, "trial_params": null }