Download checkpoint-5209/trainer_state.json from ChavyvAkvar/pathfox-baseline-1-epoch: direct link, hf CLI and curl.
- Browser
- Download file 18.8 kB
-
https://huggingface.co/ChavyvAkvar/pathfox-baseline-1-epoch/resolve/main/checkpoint-5209/trainer_state.json
- Command line
-
hf download hf://ChavyvAkvar/pathfox-baseline-1-epoch/checkpoint-5209/trainer_state.json
-
curl -L -o trainer_state.json https://huggingface.co/ChavyvAkvar/pathfox-baseline-1-epoch/resolve/main/checkpoint-5209/trainer_state.json
18.8 kB
| { | |
| "best_global_step": null, | |
| "best_metric": null, | |
| "best_model_checkpoint": null, | |
| "epoch": 1.0, | |
| "eval_steps": 500, | |
| "global_step": 5209, | |
| "is_hyper_param_search": false, | |
| "is_local_process_zero": true, | |
| "is_world_process_zero": true, | |
| "log_history": [ | |
| { | |
| "epoch": 0.009599692809830085, | |
| "grad_norm": 7.563427925109863, | |
| "learning_rate": 1.47e-05, | |
| "loss": 20.2245, | |
| "step": 50 | |
| }, | |
| { | |
| "epoch": 0.01919938561966017, | |
| "grad_norm": 29.188539505004883, | |
| "learning_rate": 2.97e-05, | |
| "loss": 16.0616, | |
| "step": 100 | |
| }, | |
| { | |
| "epoch": 0.028799078429490255, | |
| "grad_norm": 38.129608154296875, | |
| "learning_rate": 4.4699999999999996e-05, | |
| "loss": 12.7127, | |
| "step": 150 | |
| }, | |
| { | |
| "epoch": 0.03839877123932034, | |
| "grad_norm": 27.457260131835938, | |
| "learning_rate": 5.97e-05, | |
| "loss": 9.8971, | |
| "step": 200 | |
| }, | |
| { | |
| "epoch": 0.047998464049150424, | |
| "grad_norm": 29.130887985229492, | |
| "learning_rate": 7.47e-05, | |
| "loss": 7.5544, | |
| "step": 250 | |
| }, | |
| { | |
| "epoch": 0.05759815685898051, | |
| "grad_norm": 22.884220123291016, | |
| "learning_rate": 8.969999999999998e-05, | |
| "loss": 5.7283, | |
| "step": 300 | |
| }, | |
| { | |
| "epoch": 0.0671978496688106, | |
| "grad_norm": 14.968835830688477, | |
| "learning_rate": 0.00010469999999999998, | |
| "loss": 4.1701, | |
| "step": 350 | |
| }, | |
| { | |
| "epoch": 0.07679754247864068, | |
| "grad_norm": 7.887697219848633, | |
| "learning_rate": 0.0001197, | |
| "loss": 3.0776, | |
| "step": 400 | |
| }, | |
| { | |
| "epoch": 0.08639723528847076, | |
| "grad_norm": 5.031055927276611, | |
| "learning_rate": 0.0001347, | |
| "loss": 2.3821, | |
| "step": 450 | |
| }, | |
| { | |
| "epoch": 0.09599692809830085, | |
| "grad_norm": 6.178770542144775, | |
| "learning_rate": 0.00014969999999999998, | |
| "loss": 2.005, | |
| "step": 500 | |
| }, | |
| { | |
| "epoch": 0.10559662090813093, | |
| "grad_norm": 5.43004035949707, | |
| "learning_rate": 0.0001647, | |
| "loss": 1.7496, | |
| "step": 550 | |
| }, | |
| { | |
| "epoch": 0.11519631371796102, | |
| "grad_norm": 6.949675559997559, | |
| "learning_rate": 0.00017969999999999998, | |
| "loss": 1.57, | |
| "step": 600 | |
| }, | |
| { | |
| "epoch": 0.12479600652779112, | |
| "grad_norm": 4.977025508880615, | |
| "learning_rate": 0.0001947, | |
| "loss": 1.4294, | |
| "step": 650 | |
| }, | |
| { | |
| "epoch": 0.1343956993376212, | |
| "grad_norm": 9.961807250976562, | |
| "learning_rate": 0.00020969999999999997, | |
| "loss": 1.3506, | |
| "step": 700 | |
| }, | |
| { | |
| "epoch": 0.14399539214745127, | |
| "grad_norm": 3.297438383102417, | |
| "learning_rate": 0.0002247, | |
| "loss": 1.2624, | |
| "step": 750 | |
| }, | |
| { | |
| "epoch": 0.15359508495728136, | |
| "grad_norm": 3.7409021854400635, | |
| "learning_rate": 0.0002397, | |
| "loss": 1.1136, | |
| "step": 800 | |
| }, | |
| { | |
| "epoch": 0.16319477776711144, | |
| "grad_norm": 2.896446466445923, | |
| "learning_rate": 0.00025469999999999996, | |
| "loss": 0.9954, | |
| "step": 850 | |
| }, | |
| { | |
| "epoch": 0.17279447057694153, | |
| "grad_norm": 4.742156982421875, | |
| "learning_rate": 0.0002697, | |
| "loss": 0.9425, | |
| "step": 900 | |
| }, | |
| { | |
| "epoch": 0.1823941633867716, | |
| "grad_norm": 2.8668880462646484, | |
| "learning_rate": 0.0002847, | |
| "loss": 0.8012, | |
| "step": 950 | |
| }, | |
| { | |
| "epoch": 0.1919938561966017, | |
| "grad_norm": 3.0938684940338135, | |
| "learning_rate": 0.00029969999999999997, | |
| "loss": 0.6747, | |
| "step": 1000 | |
| }, | |
| { | |
| "epoch": 0.20159354900643178, | |
| "grad_norm": 2.447160005569458, | |
| "learning_rate": 0.0002965074839629365, | |
| "loss": 0.5663, | |
| "step": 1050 | |
| }, | |
| { | |
| "epoch": 0.21119324181626187, | |
| "grad_norm": 1.987984299659729, | |
| "learning_rate": 0.000292943692088382, | |
| "loss": 0.4893, | |
| "step": 1100 | |
| }, | |
| { | |
| "epoch": 0.22079293462609195, | |
| "grad_norm": 1.7457754611968994, | |
| "learning_rate": 0.0002893799002138275, | |
| "loss": 0.438, | |
| "step": 1150 | |
| }, | |
| { | |
| "epoch": 0.23039262743592204, | |
| "grad_norm": 1.5488331317901611, | |
| "learning_rate": 0.000285816108339273, | |
| "loss": 0.3947, | |
| "step": 1200 | |
| }, | |
| { | |
| "epoch": 0.23999232024575212, | |
| "grad_norm": 2.7451088428497314, | |
| "learning_rate": 0.00028225231646471845, | |
| "loss": 0.3685, | |
| "step": 1250 | |
| }, | |
| { | |
| "epoch": 0.24959201305558223, | |
| "grad_norm": 1.4951989650726318, | |
| "learning_rate": 0.0002786885245901639, | |
| "loss": 0.3416, | |
| "step": 1300 | |
| }, | |
| { | |
| "epoch": 0.2591917058654123, | |
| "grad_norm": 1.9799026250839233, | |
| "learning_rate": 0.0002751247327156094, | |
| "loss": 0.3128, | |
| "step": 1350 | |
| }, | |
| { | |
| "epoch": 0.2687913986752424, | |
| "grad_norm": 1.9234377145767212, | |
| "learning_rate": 0.00027156094084105487, | |
| "loss": 0.2892, | |
| "step": 1400 | |
| }, | |
| { | |
| "epoch": 0.27839109148507246, | |
| "grad_norm": 1.1122220754623413, | |
| "learning_rate": 0.00026799714896650034, | |
| "loss": 0.2641, | |
| "step": 1450 | |
| }, | |
| { | |
| "epoch": 0.28799078429490255, | |
| "grad_norm": 1.6626770496368408, | |
| "learning_rate": 0.0002644333570919458, | |
| "loss": 0.2503, | |
| "step": 1500 | |
| }, | |
| { | |
| "epoch": 0.29759047710473263, | |
| "grad_norm": 0.8336459398269653, | |
| "learning_rate": 0.0002608695652173913, | |
| "loss": 0.2558, | |
| "step": 1550 | |
| }, | |
| { | |
| "epoch": 0.3071901699145627, | |
| "grad_norm": 0.6865236163139343, | |
| "learning_rate": 0.00025730577334283675, | |
| "loss": 0.2341, | |
| "step": 1600 | |
| }, | |
| { | |
| "epoch": 0.3167898627243928, | |
| "grad_norm": 0.6718222498893738, | |
| "learning_rate": 0.0002537419814682822, | |
| "loss": 0.2129, | |
| "step": 1650 | |
| }, | |
| { | |
| "epoch": 0.3263895555342229, | |
| "grad_norm": 1.026585340499878, | |
| "learning_rate": 0.0002501781895937277, | |
| "loss": 0.2072, | |
| "step": 1700 | |
| }, | |
| { | |
| "epoch": 0.33598924834405297, | |
| "grad_norm": 0.6105709671974182, | |
| "learning_rate": 0.00024661439771917317, | |
| "loss": 0.2088, | |
| "step": 1750 | |
| }, | |
| { | |
| "epoch": 0.34558894115388306, | |
| "grad_norm": 0.6215556263923645, | |
| "learning_rate": 0.00024305060584461867, | |
| "loss": 0.1949, | |
| "step": 1800 | |
| }, | |
| { | |
| "epoch": 0.35518863396371314, | |
| "grad_norm": 0.8784211874008179, | |
| "learning_rate": 0.00023948681397006414, | |
| "loss": 0.1872, | |
| "step": 1850 | |
| }, | |
| { | |
| "epoch": 0.3647883267735432, | |
| "grad_norm": 0.6036493182182312, | |
| "learning_rate": 0.0002359230220955096, | |
| "loss": 0.1776, | |
| "step": 1900 | |
| }, | |
| { | |
| "epoch": 0.3743880195833733, | |
| "grad_norm": 1.1696622371673584, | |
| "learning_rate": 0.00023235923022095505, | |
| "loss": 0.1665, | |
| "step": 1950 | |
| }, | |
| { | |
| "epoch": 0.3839877123932034, | |
| "grad_norm": 0.624737560749054, | |
| "learning_rate": 0.00022879543834640053, | |
| "loss": 0.1595, | |
| "step": 2000 | |
| }, | |
| { | |
| "epoch": 0.3935874052030335, | |
| "grad_norm": 0.6369844079017639, | |
| "learning_rate": 0.00022523164647184602, | |
| "loss": 0.1548, | |
| "step": 2050 | |
| }, | |
| { | |
| "epoch": 0.40318709801286357, | |
| "grad_norm": 0.4880986511707306, | |
| "learning_rate": 0.0002216678545972915, | |
| "loss": 0.1525, | |
| "step": 2100 | |
| }, | |
| { | |
| "epoch": 0.41278679082269365, | |
| "grad_norm": 0.6220801472663879, | |
| "learning_rate": 0.00021810406272273697, | |
| "loss": 0.1483, | |
| "step": 2150 | |
| }, | |
| { | |
| "epoch": 0.42238648363252373, | |
| "grad_norm": 0.6783862709999084, | |
| "learning_rate": 0.00021454027084818244, | |
| "loss": 0.1469, | |
| "step": 2200 | |
| }, | |
| { | |
| "epoch": 0.4319861764423538, | |
| "grad_norm": 0.9673178195953369, | |
| "learning_rate": 0.00021097647897362794, | |
| "loss": 0.1472, | |
| "step": 2250 | |
| }, | |
| { | |
| "epoch": 0.4415858692521839, | |
| "grad_norm": 0.5814102292060852, | |
| "learning_rate": 0.0002074126870990734, | |
| "loss": 0.1451, | |
| "step": 2300 | |
| }, | |
| { | |
| "epoch": 0.451185562062014, | |
| "grad_norm": 2.7569072246551514, | |
| "learning_rate": 0.00020384889522451888, | |
| "loss": 0.1535, | |
| "step": 2350 | |
| }, | |
| { | |
| "epoch": 0.4607852548718441, | |
| "grad_norm": 0.7196153402328491, | |
| "learning_rate": 0.00020028510334996435, | |
| "loss": 0.1564, | |
| "step": 2400 | |
| }, | |
| { | |
| "epoch": 0.47038494768167416, | |
| "grad_norm": 0.44885897636413574, | |
| "learning_rate": 0.0001967213114754098, | |
| "loss": 0.1474, | |
| "step": 2450 | |
| }, | |
| { | |
| "epoch": 0.47998464049150424, | |
| "grad_norm": 0.715428352355957, | |
| "learning_rate": 0.00019315751960085527, | |
| "loss": 0.1366, | |
| "step": 2500 | |
| }, | |
| { | |
| "epoch": 0.4895843333013344, | |
| "grad_norm": 0.33587202429771423, | |
| "learning_rate": 0.00018959372772630077, | |
| "loss": 0.1338, | |
| "step": 2550 | |
| }, | |
| { | |
| "epoch": 0.49918402611116447, | |
| "grad_norm": 0.3589983284473419, | |
| "learning_rate": 0.00018602993585174624, | |
| "loss": 0.1256, | |
| "step": 2600 | |
| }, | |
| { | |
| "epoch": 0.5087837189209945, | |
| "grad_norm": 0.2169811725616455, | |
| "learning_rate": 0.0001824661439771917, | |
| "loss": 0.1286, | |
| "step": 2650 | |
| }, | |
| { | |
| "epoch": 0.5183834117308246, | |
| "grad_norm": 0.4184163510799408, | |
| "learning_rate": 0.00017890235210263718, | |
| "loss": 0.1246, | |
| "step": 2700 | |
| }, | |
| { | |
| "epoch": 0.5279831045406547, | |
| "grad_norm": 0.33475199341773987, | |
| "learning_rate": 0.00017533856022808268, | |
| "loss": 0.1249, | |
| "step": 2750 | |
| }, | |
| { | |
| "epoch": 0.5375827973504848, | |
| "grad_norm": 0.5484436750411987, | |
| "learning_rate": 0.00017177476835352815, | |
| "loss": 0.125, | |
| "step": 2800 | |
| }, | |
| { | |
| "epoch": 0.5471824901603148, | |
| "grad_norm": 0.34854593873023987, | |
| "learning_rate": 0.00016821097647897363, | |
| "loss": 0.1277, | |
| "step": 2850 | |
| }, | |
| { | |
| "epoch": 0.5567821829701449, | |
| "grad_norm": 0.3453426659107208, | |
| "learning_rate": 0.0001646471846044191, | |
| "loss": 0.1271, | |
| "step": 2900 | |
| }, | |
| { | |
| "epoch": 0.566381875779975, | |
| "grad_norm": 0.34477025270462036, | |
| "learning_rate": 0.00016108339272986454, | |
| "loss": 0.1211, | |
| "step": 2950 | |
| }, | |
| { | |
| "epoch": 0.5759815685898051, | |
| "grad_norm": 0.2906150221824646, | |
| "learning_rate": 0.00015751960085531, | |
| "loss": 0.124, | |
| "step": 3000 | |
| }, | |
| { | |
| "epoch": 0.5855812613996352, | |
| "grad_norm": 0.2288372963666916, | |
| "learning_rate": 0.0001539558089807555, | |
| "loss": 0.1189, | |
| "step": 3050 | |
| }, | |
| { | |
| "epoch": 0.5951809542094653, | |
| "grad_norm": 0.3227977752685547, | |
| "learning_rate": 0.00015039201710620098, | |
| "loss": 0.1227, | |
| "step": 3100 | |
| }, | |
| { | |
| "epoch": 0.6047806470192953, | |
| "grad_norm": 0.40704479813575745, | |
| "learning_rate": 0.00014682822523164645, | |
| "loss": 0.1202, | |
| "step": 3150 | |
| }, | |
| { | |
| "epoch": 0.6143803398291254, | |
| "grad_norm": 0.3058551847934723, | |
| "learning_rate": 0.00014326443335709193, | |
| "loss": 0.1178, | |
| "step": 3200 | |
| }, | |
| { | |
| "epoch": 0.6239800326389555, | |
| "grad_norm": 0.2437949925661087, | |
| "learning_rate": 0.00013970064148253743, | |
| "loss": 0.1156, | |
| "step": 3250 | |
| }, | |
| { | |
| "epoch": 0.6335797254487856, | |
| "grad_norm": 0.24965952336788177, | |
| "learning_rate": 0.0001361368496079829, | |
| "loss": 0.1131, | |
| "step": 3300 | |
| }, | |
| { | |
| "epoch": 0.6431794182586157, | |
| "grad_norm": 0.4167115092277527, | |
| "learning_rate": 0.00013257305773342834, | |
| "loss": 0.115, | |
| "step": 3350 | |
| }, | |
| { | |
| "epoch": 0.6527791110684458, | |
| "grad_norm": 0.30340147018432617, | |
| "learning_rate": 0.00012900926585887384, | |
| "loss": 0.1131, | |
| "step": 3400 | |
| }, | |
| { | |
| "epoch": 0.6623788038782759, | |
| "grad_norm": 0.43558454513549805, | |
| "learning_rate": 0.0001254454739843193, | |
| "loss": 0.113, | |
| "step": 3450 | |
| }, | |
| { | |
| "epoch": 0.6719784966881059, | |
| "grad_norm": 0.30567434430122375, | |
| "learning_rate": 0.00012188168210976478, | |
| "loss": 0.1142, | |
| "step": 3500 | |
| }, | |
| { | |
| "epoch": 0.681578189497936, | |
| "grad_norm": 0.28799161314964294, | |
| "learning_rate": 0.00011831789023521026, | |
| "loss": 0.1111, | |
| "step": 3550 | |
| }, | |
| { | |
| "epoch": 0.6911778823077661, | |
| "grad_norm": 0.28393957018852234, | |
| "learning_rate": 0.00011475409836065573, | |
| "loss": 0.1117, | |
| "step": 3600 | |
| }, | |
| { | |
| "epoch": 0.7007775751175962, | |
| "grad_norm": 0.27800044417381287, | |
| "learning_rate": 0.0001111903064861012, | |
| "loss": 0.111, | |
| "step": 3650 | |
| }, | |
| { | |
| "epoch": 0.7103772679274263, | |
| "grad_norm": 0.19541418552398682, | |
| "learning_rate": 0.00010762651461154668, | |
| "loss": 0.1092, | |
| "step": 3700 | |
| }, | |
| { | |
| "epoch": 0.7199769607372564, | |
| "grad_norm": 0.24023820459842682, | |
| "learning_rate": 0.00010406272273699216, | |
| "loss": 0.1099, | |
| "step": 3750 | |
| }, | |
| { | |
| "epoch": 0.7295766535470865, | |
| "grad_norm": 0.259033203125, | |
| "learning_rate": 0.00010049893086243763, | |
| "loss": 0.109, | |
| "step": 3800 | |
| }, | |
| { | |
| "epoch": 0.7391763463569165, | |
| "grad_norm": 0.25941726565361023, | |
| "learning_rate": 9.69351389878831e-05, | |
| "loss": 0.106, | |
| "step": 3850 | |
| }, | |
| { | |
| "epoch": 0.7487760391667466, | |
| "grad_norm": 0.15513530373573303, | |
| "learning_rate": 9.337134711332857e-05, | |
| "loss": 0.1056, | |
| "step": 3900 | |
| }, | |
| { | |
| "epoch": 0.7583757319765767, | |
| "grad_norm": 0.2611596882343292, | |
| "learning_rate": 8.980755523877404e-05, | |
| "loss": 0.1065, | |
| "step": 3950 | |
| }, | |
| { | |
| "epoch": 0.7679754247864068, | |
| "grad_norm": 0.2952909469604492, | |
| "learning_rate": 8.624376336421953e-05, | |
| "loss": 0.1066, | |
| "step": 4000 | |
| }, | |
| { | |
| "epoch": 0.7775751175962369, | |
| "grad_norm": 0.251472532749176, | |
| "learning_rate": 8.2679971489665e-05, | |
| "loss": 0.1082, | |
| "step": 4050 | |
| }, | |
| { | |
| "epoch": 0.787174810406067, | |
| "grad_norm": 0.1387886255979538, | |
| "learning_rate": 7.911617961511047e-05, | |
| "loss": 0.1055, | |
| "step": 4100 | |
| }, | |
| { | |
| "epoch": 0.796774503215897, | |
| "grad_norm": 0.28730371594429016, | |
| "learning_rate": 7.555238774055594e-05, | |
| "loss": 0.1071, | |
| "step": 4150 | |
| }, | |
| { | |
| "epoch": 0.8063741960257271, | |
| "grad_norm": 0.12195300310850143, | |
| "learning_rate": 7.198859586600141e-05, | |
| "loss": 0.1048, | |
| "step": 4200 | |
| }, | |
| { | |
| "epoch": 0.8159738888355572, | |
| "grad_norm": 0.22190533578395844, | |
| "learning_rate": 6.84248039914469e-05, | |
| "loss": 0.1025, | |
| "step": 4250 | |
| }, | |
| { | |
| "epoch": 0.8255735816453873, | |
| "grad_norm": 0.1989029198884964, | |
| "learning_rate": 6.486101211689237e-05, | |
| "loss": 0.1015, | |
| "step": 4300 | |
| }, | |
| { | |
| "epoch": 0.8351732744552174, | |
| "grad_norm": 0.1682548075914383, | |
| "learning_rate": 6.129722024233784e-05, | |
| "loss": 0.1033, | |
| "step": 4350 | |
| }, | |
| { | |
| "epoch": 0.8447729672650475, | |
| "grad_norm": 0.13227292895317078, | |
| "learning_rate": 5.7733428367783314e-05, | |
| "loss": 0.1038, | |
| "step": 4400 | |
| }, | |
| { | |
| "epoch": 0.8543726600748776, | |
| "grad_norm": 0.19235797226428986, | |
| "learning_rate": 5.416963649322879e-05, | |
| "loss": 0.1034, | |
| "step": 4450 | |
| }, | |
| { | |
| "epoch": 0.8639723528847076, | |
| "grad_norm": 0.1321435123682022, | |
| "learning_rate": 5.060584461867427e-05, | |
| "loss": 0.1029, | |
| "step": 4500 | |
| }, | |
| { | |
| "epoch": 0.8735720456945377, | |
| "grad_norm": 0.1909589022397995, | |
| "learning_rate": 4.7042052744119735e-05, | |
| "loss": 0.1045, | |
| "step": 4550 | |
| }, | |
| { | |
| "epoch": 0.8831717385043678, | |
| "grad_norm": 0.15285654366016388, | |
| "learning_rate": 4.3478260869565214e-05, | |
| "loss": 0.1032, | |
| "step": 4600 | |
| }, | |
| { | |
| "epoch": 0.8927714313141979, | |
| "grad_norm": 0.1674521118402481, | |
| "learning_rate": 3.9914468995010685e-05, | |
| "loss": 0.1023, | |
| "step": 4650 | |
| }, | |
| { | |
| "epoch": 0.902371124124028, | |
| "grad_norm": 0.11908634752035141, | |
| "learning_rate": 3.6350677120456164e-05, | |
| "loss": 0.1021, | |
| "step": 4700 | |
| }, | |
| { | |
| "epoch": 0.9119708169338581, | |
| "grad_norm": 0.1627221256494522, | |
| "learning_rate": 3.2786885245901635e-05, | |
| "loss": 0.1019, | |
| "step": 4750 | |
| }, | |
| { | |
| "epoch": 0.9215705097436881, | |
| "grad_norm": 0.17971652746200562, | |
| "learning_rate": 2.922309337134711e-05, | |
| "loss": 0.1003, | |
| "step": 4800 | |
| }, | |
| { | |
| "epoch": 0.9311702025535182, | |
| "grad_norm": 0.1855141818523407, | |
| "learning_rate": 2.5659301496792585e-05, | |
| "loss": 0.1036, | |
| "step": 4850 | |
| }, | |
| { | |
| "epoch": 0.9407698953633483, | |
| "grad_norm": 0.13592451810836792, | |
| "learning_rate": 2.209550962223806e-05, | |
| "loss": 0.101, | |
| "step": 4900 | |
| }, | |
| { | |
| "epoch": 0.9503695881731784, | |
| "grad_norm": 0.18601875007152557, | |
| "learning_rate": 1.8531717747683532e-05, | |
| "loss": 0.1019, | |
| "step": 4950 | |
| }, | |
| { | |
| "epoch": 0.9599692809830085, | |
| "grad_norm": 0.15403127670288086, | |
| "learning_rate": 1.4967925873129009e-05, | |
| "loss": 0.0992, | |
| "step": 5000 | |
| }, | |
| { | |
| "epoch": 0.9695689737928387, | |
| "grad_norm": 0.13198044896125793, | |
| "learning_rate": 1.1404133998574482e-05, | |
| "loss": 0.1023, | |
| "step": 5050 | |
| }, | |
| { | |
| "epoch": 0.9791686666026688, | |
| "grad_norm": 0.139701709151268, | |
| "learning_rate": 7.840342124019957e-06, | |
| "loss": 0.1015, | |
| "step": 5100 | |
| }, | |
| { | |
| "epoch": 0.9887683594124989, | |
| "grad_norm": 0.1343325823545456, | |
| "learning_rate": 4.276550249465431e-06, | |
| "loss": 0.1036, | |
| "step": 5150 | |
| }, | |
| { | |
| "epoch": 0.9983680522223289, | |
| "grad_norm": 0.1440255045890808, | |
| "learning_rate": 7.127583749109051e-07, | |
| "loss": 0.1022, | |
| "step": 5200 | |
| } | |
| ], | |
| "logging_steps": 50, | |
| "max_steps": 5209, | |
| "num_input_tokens_seen": 0, | |
| "num_train_epochs": 1, | |
| "save_steps": 2000, | |
| "stateful_callbacks": { | |
| "TrainerControl": { | |
| "args": { | |
| "should_epoch_stop": false, | |
| "should_evaluate": false, | |
| "should_log": false, | |
| "should_save": true, | |
| "should_training_stop": true | |
| }, | |
| "attributes": {} | |
| } | |
| }, | |
| "total_flos": 1.1425482385706189e+17, | |
| "train_batch_size": 16, | |
| "trial_name": null, | |
| "trial_params": null | |
| } | |