avsolatorio/data-use-unsloth-phi-3.5-data-datause-climatechage-iclr461-100epochs-1736824522-lora
{"epoch": 41.0, "global_step": 33620, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.9029145756309676e+19, "log_history": [{"loss": 1.3163, "grad_norm": 0.756934404373169, "learning_rate": 1.2113720642768852e-05, "epoch": 40.12206286237412, "step": 32900}, {"loss": 1.3048, "grad_norm": 0.7757614850997925, "learning_rate": 1.2088998763906057e-05, "epoch": 40.244125724748244, "step": 33000}, {"loss": 1.3198, "grad_norm": 0.8720489740371704, "learning_rate": 1.2064276885043264e-05, "epoch": 40.366188587122366, "step": 33100}, {"loss": 1.3182, "grad_norm": 0.8135008811950684, "learning_rate": 1.203955500618047e-05, "epoch": 40.48825144949649, "step": 33200}, {"loss": 1.2975, "grad_norm": 0.982002854347229, "learning_rate": 1.2014833127317677e-05, "epoch": 40.61031431187061, "step": 33300}, {"loss": 1.3096, "grad_norm": 0.9260008931159973, "learning_rate": 1.1990111248454883e-05, "epoch": 40.73237717424474, "step": 33400}, {"loss": 1.2911, "grad_norm": 0.9837338924407959, "learning_rate": 1.196538936959209e-05, "epoch": 40.85444003661886, "step": 33500}, {"loss": 1.3091, "grad_norm": 0.820532500743866, "learning_rate": 1.1940667490729296e-05, "epoch": 40.97650289899298, "step": 33600}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 40.0, "global_step": 32800, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.857542574510646e+19, "log_history": [{"loss": 1.3202, "grad_norm": 0.8156407475471497, "learning_rate": 1.2336217552533993e-05, "epoch": 39.02441257247482, "step": 32000}, {"loss": 1.3085, "grad_norm": 0.88544100522995, "learning_rate": 1.23114956736712e-05, "epoch": 39.146475434848945, "step": 32100}, {"loss": 1.3314, "grad_norm": 0.7666390538215637, "learning_rate": 1.2286773794808406e-05, "epoch": 39.26853829722307, "step": 32200}, {"loss": 1.3065, "grad_norm": 0.906538724899292, "learning_rate": 1.2262051915945612e-05, "epoch": 39.390601159597196, "step": 32300}, {"loss": 1.3128, "grad_norm": 0.7657581567764282, "learning_rate": 1.223733003708282e-05, "epoch": 39.51266402197132, "step": 32400}, {"loss": 1.3277, "grad_norm": 1.096739411354065, "learning_rate": 1.2212608158220024e-05, "epoch": 39.63472688434544, "step": 32500}, {"loss": 1.3164, "grad_norm": 0.8237380385398865, "learning_rate": 1.2187886279357233e-05, "epoch": 39.75678974671956, "step": 32600}, {"loss": 1.3032, "grad_norm": 0.9224328994750977, "learning_rate": 1.2163164400494439e-05, "epoch": 39.878852609093684, "step": 32700}, {"loss": 1.2899, "grad_norm": 2.1877095699310303, "learning_rate": 1.2138442521631646e-05, "epoch": 40.0, "step": 32800}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 39.0, "global_step": 31980, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.8065736647169546e+19, "log_history": [{"loss": 1.3215, "grad_norm": 0.9206139445304871, "learning_rate": 1.2533992583436344e-05, "epoch": 38.048825144949646, "step": 31200}, {"loss": 1.3147, "grad_norm": 0.7489995360374451, "learning_rate": 1.2509270704573549e-05, "epoch": 38.170888007323775, "step": 31300}, {"loss": 1.3177, "grad_norm": 0.8376150727272034, "learning_rate": 1.2484548825710756e-05, "epoch": 38.2929508696979, "step": 31400}, {"loss": 1.2961, "grad_norm": 0.8982778191566467, "learning_rate": 1.2459826946847962e-05, "epoch": 38.41501373207202, "step": 31500}, {"loss": 1.3314, "grad_norm": 0.8565440773963928, "learning_rate": 1.2435105067985168e-05, "epoch": 38.53707659444614, "step": 31600}, {"loss": 1.3254, "grad_norm": 0.8229475021362305, "learning_rate": 1.2410383189122375e-05, "epoch": 38.65913945682026, "step": 31700}, {"loss": 1.3129, "grad_norm": 0.9277706146240234, "learning_rate": 1.238566131025958e-05, "epoch": 38.781202319194385, "step": 31800}, {"loss": 1.3132, "grad_norm": 0.8812633752822876, "learning_rate": 1.2360939431396788e-05, "epoch": 38.90326518156851, "step": 31900}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 38.0, "global_step": 31160, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.7611272657570724e+19, "log_history": [{"loss": 1.321, "grad_norm": 0.7953139543533325, "learning_rate": 1.273176761433869e-05, "epoch": 37.073237717424476, "step": 30400}, {"loss": 1.3113, "grad_norm": 0.8607418537139893, "learning_rate": 1.2707045735475898e-05, "epoch": 37.1953005797986, "step": 30500}, {"loss": 1.3148, "grad_norm": 0.7135468125343323, "learning_rate": 1.2682323856613103e-05, "epoch": 37.31736344217272, "step": 30600}, {"loss": 1.2896, "grad_norm": 0.7779369950294495, "learning_rate": 1.2657601977750309e-05, "epoch": 37.43942630454684, "step": 30700}, {"loss": 1.3501, "grad_norm": 0.820813775062561, "learning_rate": 1.2632880098887516e-05, "epoch": 37.561489166920964, "step": 30800}, {"loss": 1.3346, "grad_norm": 0.7352159023284912, "learning_rate": 1.2608158220024722e-05, "epoch": 37.683552029295086, "step": 30900}, {"loss": 1.3143, "grad_norm": 0.8640016913414001, "learning_rate": 1.2583436341161929e-05, "epoch": 37.80561489166921, "step": 31000}, {"loss": 1.3279, "grad_norm": 0.7870578169822693, "learning_rate": 1.2558714462299135e-05, "epoch": 37.92767775404333, "step": 31100}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 37.0, "global_step": 30340, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.7159337595623813e+19, "log_history": [{"loss": 1.3006, "grad_norm": 0.8262194395065308, "learning_rate": 1.292954264524104e-05, "epoch": 36.0976502898993, "step": 29600}, {"loss": 1.3233, "grad_norm": 0.7193596959114075, "learning_rate": 1.2904820766378245e-05, "epoch": 36.21971315227342, "step": 29700}, {"loss": 1.3215, "grad_norm": 1.3133962154388428, "learning_rate": 1.2880098887515454e-05, "epoch": 36.34177601464754, "step": 29800}, {"loss": 1.3108, "grad_norm": 0.7914173603057861, "learning_rate": 1.2855377008652658e-05, "epoch": 36.463838877021665, "step": 29900}, {"loss": 1.3344, "grad_norm": 0.6829717755317688, "learning_rate": 1.2830655129789863e-05, "epoch": 36.58590173939579, "step": 30000}, {"loss": 1.345, "grad_norm": 0.7118475437164307, "learning_rate": 1.2805933250927072e-05, "epoch": 36.70796460176991, "step": 30100}, {"loss": 1.3309, "grad_norm": 0.812768816947937, "learning_rate": 1.2781211372064278e-05, "epoch": 36.83002746414403, "step": 30200}, {"loss": 1.3321, "grad_norm": 0.7125794291496277, "learning_rate": 1.2756489493201485e-05, "epoch": 36.95209032651816, "step": 30300}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 36.0, "global_step": 29520, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.6705433143341527e+19, "log_history": [{"loss": 1.3215, "grad_norm": 0.6374627351760864, "learning_rate": 1.3127317676143388e-05, "epoch": 35.12206286237412, "step": 28800}, {"loss": 1.3281, "grad_norm": 1.0293234586715698, "learning_rate": 1.3102595797280595e-05, "epoch": 35.244125724748244, "step": 28900}, {"loss": 1.3214, "grad_norm": 0.7984252572059631, "learning_rate": 1.3077873918417801e-05, "epoch": 35.366188587122366, "step": 29000}, {"loss": 1.3105, "grad_norm": 0.6331897377967834, "learning_rate": 1.3053152039555008e-05, "epoch": 35.48825144949649, "step": 29100}, {"loss": 1.3484, "grad_norm": 0.6543028354644775, "learning_rate": 1.3028430160692214e-05, "epoch": 35.61031431187061, "step": 29200}, {"loss": 1.3379, "grad_norm": 0.6947120428085327, "learning_rate": 1.300370828182942e-05, "epoch": 35.73237717424474, "step": 29300}, {"loss": 1.333, "grad_norm": 0.7469784617424011, "learning_rate": 1.2978986402966627e-05, "epoch": 35.85444003661886, "step": 29400}, {"loss": 1.3473, "grad_norm": 0.6753571033477783, "learning_rate": 1.2954264524103832e-05, "epoch": 35.97650289899298, "step": 29500}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 35.0, "global_step": 28700, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.6250000928136704e+19, "log_history": [{"loss": 1.3483, "grad_norm": 0.7128898501396179, "learning_rate": 1.334981458590853e-05, "epoch": 34.02441257247482, "step": 27900}, {"loss": 1.3305, "grad_norm": 0.7221057415008545, "learning_rate": 1.3325092707045737e-05, "epoch": 34.146475434848945, "step": 28000}, {"loss": 1.3354, "grad_norm": 0.7727730870246887, "learning_rate": 1.3300370828182942e-05, "epoch": 34.26853829722307, "step": 28100}, {"loss": 1.336, "grad_norm": 0.7282364368438721, "learning_rate": 1.327564894932015e-05, "epoch": 34.390601159597196, "step": 28200}, {"loss": 1.3328, "grad_norm": 0.6162666082382202, "learning_rate": 1.3250927070457355e-05, "epoch": 34.51266402197132, "step": 28300}, {"loss": 1.3288, "grad_norm": 0.8243314027786255, "learning_rate": 1.3226205191594562e-05, "epoch": 34.63472688434544, "step": 28400}, {"loss": 1.3375, "grad_norm": 0.9031250476837158, "learning_rate": 1.3201483312731768e-05, "epoch": 34.75678974671956, "step": 28500}, {"loss": 1.3276, "grad_norm": 0.8034760355949402, "learning_rate": 1.3176761433868974e-05, "epoch": 34.878852609093684, "step": 28600}, {"loss": 1.3412, "grad_norm": 2.3968396186828613, "learning_rate": 1.3152039555006183e-05, "epoch": 35.0, "step": 28700}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 34.0, "global_step": 27880, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.5742849583645522e+19, "log_history": [{"loss": 1.3486, "grad_norm": 0.6413136124610901, "learning_rate": 1.3547589616810878e-05, "epoch": 33.048825144949646, "step": 27100}, {"loss": 1.3279, "grad_norm": 0.6214497685432434, "learning_rate": 1.3522867737948084e-05, "epoch": 33.170888007323775, "step": 27200}, {"loss": 1.3346, "grad_norm": 0.7530773878097534, "learning_rate": 1.3498145859085293e-05, "epoch": 33.2929508696979, "step": 27300}, {"loss": 1.3433, "grad_norm": 0.6536210179328918, "learning_rate": 1.3473423980222497e-05, "epoch": 33.41501373207202, "step": 27400}, {"loss": 1.3397, "grad_norm": 0.7106990814208984, "learning_rate": 1.3448702101359706e-05, "epoch": 33.53707659444614, "step": 27500}, {"loss": 1.329, "grad_norm": 1.1681946516036987, "learning_rate": 1.3423980222496911e-05, "epoch": 33.65913945682026, "step": 27600}, {"loss": 1.3424, "grad_norm": 3.5079076290130615, "learning_rate": 1.3399258343634118e-05, "epoch": 33.781202319194385, "step": 27700}, {"loss": 1.3545, "grad_norm": 0.722295880317688, "learning_rate": 1.3374536464771324e-05, "epoch": 33.90326518156851, "step": 27800}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 33.0, "global_step": 27060, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.5288010295176212e+19, "log_history": [{"loss": 1.3477, "grad_norm": 0.7199577689170837, "learning_rate": 1.3745364647713227e-05, "epoch": 32.073237717424476, "step": 26300}, {"loss": 1.3473, "grad_norm": 0.8087741732597351, "learning_rate": 1.3720642768850434e-05, "epoch": 32.1953005797986, "step": 26400}, {"loss": 1.3431, "grad_norm": 0.6292794942855835, "learning_rate": 1.369592088998764e-05, "epoch": 32.31736344217272, "step": 26500}, {"loss": 1.3389, "grad_norm": 0.7322786450386047, "learning_rate": 1.3671199011124847e-05, "epoch": 32.43942630454684, "step": 26600}, {"loss": 1.3526, "grad_norm": 0.8401209115982056, "learning_rate": 1.3646477132262053e-05, "epoch": 32.561489166920964, "step": 26700}, {"loss": 1.3411, "grad_norm": 0.6629768013954163, "learning_rate": 1.362175525339926e-05, "epoch": 32.683552029295086, "step": 26800}, {"loss": 1.3463, "grad_norm": 0.709342360496521, "learning_rate": 1.3597033374536466e-05, "epoch": 32.80561489166921, "step": 26900}, {"loss": 1.3517, "grad_norm": 0.8619031310081482, "learning_rate": 1.3572311495673671e-05, "epoch": 32.92767775404333, "step": 27000}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 32.0, "global_step": 26240, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.4833994867542518e+19, "log_history": [{"loss": 1.3547, "grad_norm": 0.8431699275970459, "learning_rate": 1.3943139678615576e-05, "epoch": 31.0976502898993, "step": 25500}, {"loss": 1.3363, "grad_norm": 0.7117323279380798, "learning_rate": 1.3918417799752781e-05, "epoch": 31.21971315227342, "step": 25600}, {"loss": 1.3508, "grad_norm": 0.6519405245780945, "learning_rate": 1.3893695920889989e-05, "epoch": 31.341776014647543, "step": 25700}, {"loss": 1.3565, "grad_norm": 0.6814839243888855, "learning_rate": 1.3868974042027194e-05, "epoch": 31.463838877021665, "step": 25800}, {"loss": 1.3589, "grad_norm": 0.7914370894432068, "learning_rate": 1.3844252163164401e-05, "epoch": 31.585901739395787, "step": 25900}, {"loss": 1.3594, "grad_norm": 0.6821210384368896, "learning_rate": 1.3819530284301607e-05, "epoch": 31.707964601769913, "step": 26000}, {"loss": 1.3709, "grad_norm": 0.8016797304153442, "learning_rate": 1.3794808405438816e-05, "epoch": 31.830027464144035, "step": 26100}, {"loss": 1.3342, "grad_norm": 0.7453998327255249, "learning_rate": 1.3770086526576022e-05, "epoch": 31.952090326518157, "step": 26200}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 31.0, "global_step": 25420, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.4381654201831025e+19, "log_history": [{"loss": 1.3588, "grad_norm": 0.6830438375473022, "learning_rate": 1.4140914709517926e-05, "epoch": 30.122062862374122, "step": 24700}, {"loss": 1.3427, "grad_norm": 0.7066377997398376, "learning_rate": 1.4116192830655132e-05, "epoch": 30.244125724748244, "step": 24800}, {"loss": 1.3495, "grad_norm": 0.7590414881706238, "learning_rate": 1.4091470951792336e-05, "epoch": 30.36618858712237, "step": 24900}, {"loss": 1.3741, "grad_norm": 0.7908617258071899, "learning_rate": 1.4066749072929545e-05, "epoch": 30.48825144949649, "step": 25000}, {"loss": 1.3619, "grad_norm": 0.7022337317466736, "learning_rate": 1.404202719406675e-05, "epoch": 30.610314311870614, "step": 25100}, {"loss": 1.339, "grad_norm": 0.8073809146881104, "learning_rate": 1.4017305315203957e-05, "epoch": 30.732377174244736, "step": 25200}, {"loss": 1.3502, "grad_norm": 0.673373818397522, "learning_rate": 1.3992583436341163e-05, "epoch": 30.854440036618858, "step": 25300}, {"loss": 1.3593, "grad_norm": 0.6519209146499634, "learning_rate": 1.396786155747837e-05, "epoch": 30.976502898992983, "step": 25400}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 30.0, "global_step": 24600, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.3925340487953377e+19, "log_history": [{"loss": 1.3692, "grad_norm": 0.5433763861656189, "learning_rate": 1.4363411619283068e-05, "epoch": 29.024412572474823, "step": 23800}, {"loss": 1.3486, "grad_norm": 0.7373917698860168, "learning_rate": 1.4338689740420273e-05, "epoch": 29.14647543484895, "step": 23900}, {"loss": 1.3585, "grad_norm": 0.5551890134811401, "learning_rate": 1.431396786155748e-05, "epoch": 29.26853829722307, "step": 24000}, {"loss": 1.3603, "grad_norm": 0.7008611559867859, "learning_rate": 1.4289245982694686e-05, "epoch": 29.390601159597193, "step": 24100}, {"loss": 1.3496, "grad_norm": 0.6557561755180359, "learning_rate": 1.4264524103831892e-05, "epoch": 29.512664021971315, "step": 24200}, {"loss": 1.3762, "grad_norm": 0.6356208920478821, "learning_rate": 1.4239802224969099e-05, "epoch": 29.634726884345437, "step": 24300}, {"loss": 1.3592, "grad_norm": 0.6172239184379578, "learning_rate": 1.4215080346106304e-05, "epoch": 29.756789746719562, "step": 24400}, {"loss": 1.356, "grad_norm": 0.673744261264801, "learning_rate": 1.4190358467243512e-05, "epoch": 29.878852609093684, "step": 24500}, {"loss": 1.3692, "grad_norm": 1.9929808378219604, "learning_rate": 1.4165636588380717e-05, "epoch": 30.0, "step": 24600}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 29.0, "global_step": 23780, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.3418366920166339e+19, "log_history": [{"loss": 1.3482, "grad_norm": 0.5828624963760376, "learning_rate": 1.4561186650185415e-05, "epoch": 28.04882514494965, "step": 23000}, {"loss": 1.3553, "grad_norm": 0.5839941501617432, "learning_rate": 1.4536464771322622e-05, "epoch": 28.17088800732377, "step": 23100}, {"loss": 1.3423, "grad_norm": 0.5885896682739258, "learning_rate": 1.4511742892459828e-05, "epoch": 28.292950869697894, "step": 23200}, {"loss": 1.3664, "grad_norm": 0.7272441387176514, "learning_rate": 1.4487021013597033e-05, "epoch": 28.415013732072016, "step": 23300}, {"loss": 1.3656, "grad_norm": 0.7931562662124634, "learning_rate": 1.446229913473424e-05, "epoch": 28.53707659444614, "step": 23400}, {"loss": 1.3688, "grad_norm": 0.7180541157722473, "learning_rate": 1.4437577255871446e-05, "epoch": 28.659139456820263, "step": 23500}, {"loss": 1.3777, "grad_norm": 0.7169442772865295, "learning_rate": 1.4412855377008655e-05, "epoch": 28.781202319194385, "step": 23600}, {"loss": 1.387, "grad_norm": 0.5753076076507568, "learning_rate": 1.438813349814586e-05, "epoch": 28.903265181568507, "step": 23700}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 28.0, "global_step": 22960, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.2965214033760573e+19, "log_history": [{"loss": 1.3683, "grad_norm": 0.5420319437980652, "learning_rate": 1.4758961681087765e-05, "epoch": 27.073237717424472, "step": 22200}, {"loss": 1.357, "grad_norm": 0.5988544225692749, "learning_rate": 1.473423980222497e-05, "epoch": 27.195300579798598, "step": 22300}, {"loss": 1.3728, "grad_norm": 0.7745524644851685, "learning_rate": 1.4709517923362178e-05, "epoch": 27.31736344217272, "step": 22400}, {"loss": 1.3813, "grad_norm": 0.6108834743499756, "learning_rate": 1.4684796044499384e-05, "epoch": 27.439426304546842, "step": 22500}, {"loss": 1.3706, "grad_norm": 0.541451632976532, "learning_rate": 1.4660074165636589e-05, "epoch": 27.561489166920964, "step": 22600}, {"loss": 1.3687, "grad_norm": 0.6324880123138428, "learning_rate": 1.4635352286773796e-05, "epoch": 27.683552029295086, "step": 22700}, {"loss": 1.3854, "grad_norm": 0.5701207518577576, "learning_rate": 1.4610630407911002e-05, "epoch": 27.805614891669208, "step": 22800}, {"loss": 1.3678, "grad_norm": 0.6929529309272766, "learning_rate": 1.458590852904821e-05, "epoch": 27.927677754043334, "step": 22900}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 27.0, "global_step": 22140, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.25117476065263e+19, "log_history": [{"loss": 1.4027, "grad_norm": 0.6980090737342834, "learning_rate": 1.4956736711990112e-05, "epoch": 26.0976502898993, "step": 21400}, {"loss": 1.3654, "grad_norm": 0.6929275393486023, "learning_rate": 1.493201483312732e-05, "epoch": 26.21971315227342, "step": 21500}, {"loss": 1.3818, "grad_norm": 0.6196820139884949, "learning_rate": 1.4907292954264525e-05, "epoch": 26.341776014647543, "step": 21600}, {"loss": 1.3681, "grad_norm": 0.5551548004150391, "learning_rate": 1.4882571075401732e-05, "epoch": 26.463838877021665, "step": 21700}, {"loss": 1.3653, "grad_norm": 0.6336193680763245, "learning_rate": 1.4857849196538938e-05, "epoch": 26.585901739395787, "step": 21800}, {"loss": 1.391, "grad_norm": 0.6208574175834656, "learning_rate": 1.4833127317676143e-05, "epoch": 26.707964601769913, "step": 21900}, {"loss": 1.38, "grad_norm": 0.6791520714759827, "learning_rate": 1.480840543881335e-05, "epoch": 26.830027464144035, "step": 22000}, {"loss": 1.3733, "grad_norm": 0.5679318904876709, "learning_rate": 1.4783683559950556e-05, "epoch": 26.952090326518157, "step": 22100}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 26.0, "global_step": 21320, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.2058608409106872e+19, "log_history": [{"loss": 1.3853, "grad_norm": 0.5225703716278076, "learning_rate": 1.5154511742892461e-05, "epoch": 25.122062862374122, "step": 20600}, {"loss": 1.381, "grad_norm": 0.5267176628112793, "learning_rate": 1.5129789864029667e-05, "epoch": 25.244125724748244, "step": 20700}, {"loss": 1.3847, "grad_norm": 0.5324319005012512, "learning_rate": 1.5105067985166875e-05, "epoch": 25.36618858712237, "step": 20800}, {"loss": 1.393, "grad_norm": 0.7791089415550232, "learning_rate": 1.508034610630408e-05, "epoch": 25.48825144949649, "step": 20900}, {"loss": 1.3743, "grad_norm": 0.632090151309967, "learning_rate": 1.5055624227441288e-05, "epoch": 25.610314311870614, "step": 21000}, {"loss": 1.3873, "grad_norm": 0.5790057182312012, "learning_rate": 1.5030902348578494e-05, "epoch": 25.732377174244736, "step": 21100}, {"loss": 1.3761, "grad_norm": 0.6948658227920532, "learning_rate": 1.50061804697157e-05, "epoch": 25.854440036618858, "step": 21200}, {"loss": 1.3665, "grad_norm": 0.5973604321479797, "learning_rate": 1.4981458590852907e-05, "epoch": 25.976502898992983, "step": 21300}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 25.0, "global_step": 20500, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.1605605448622193e+19, "log_history": [{"loss": 1.3981, "grad_norm": 0.6444383263587952, "learning_rate": 1.5377008652657602e-05, "epoch": 24.024412572474823, "step": 19700}, {"loss": 1.3782, "grad_norm": 0.5484598278999329, "learning_rate": 1.535228677379481e-05, "epoch": 24.14647543484895, "step": 19800}, {"loss": 1.391, "grad_norm": 0.5793507099151611, "learning_rate": 1.5327564894932017e-05, "epoch": 24.26853829722307, "step": 19900}, {"loss": 1.3831, "grad_norm": 0.49974507093429565, "learning_rate": 1.530284301606922e-05, "epoch": 24.390601159597193, "step": 20000}, {"loss": 1.3822, "grad_norm": 0.6011673212051392, "learning_rate": 1.5278121137206428e-05, "epoch": 24.512664021971315, "step": 20100}, {"loss": 1.3904, "grad_norm": 0.7003011703491211, "learning_rate": 1.5253399258343635e-05, "epoch": 24.634726884345437, "step": 20200}, {"loss": 1.3721, "grad_norm": 0.6181715130805969, "learning_rate": 1.5228677379480841e-05, "epoch": 24.756789746719562, "step": 20300}, {"loss": 1.4083, "grad_norm": 0.563213050365448, "learning_rate": 1.5203955500618048e-05, "epoch": 24.878852609093684, "step": 20400}, {"loss": 1.3921, "grad_norm": 1.2892804145812988, "learning_rate": 1.5179233621755254e-05, "epoch": 25.0, "step": 20500}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 24.0, "global_step": 19680, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.1098719440815196e+19, "log_history": [{"loss": 1.4097, "grad_norm": 0.6347495913505554, "learning_rate": 1.557478368355995e-05, "epoch": 23.04882514494965, "step": 18900}, {"loss": 1.391, "grad_norm": 0.8790028095245361, "learning_rate": 1.555006180469716e-05, "epoch": 23.17088800732377, "step": 19000}, {"loss": 1.3934, "grad_norm": 0.5554202198982239, "learning_rate": 1.5525339925834362e-05, "epoch": 23.292950869697894, "step": 19100}, {"loss": 1.3892, "grad_norm": 0.5877499580383301, "learning_rate": 1.5500618046971573e-05, "epoch": 23.415013732072016, "step": 19200}, {"loss": 1.3837, "grad_norm": 0.7559616565704346, "learning_rate": 1.5475896168108777e-05, "epoch": 23.53707659444614, "step": 19300}, {"loss": 1.3959, "grad_norm": 0.5316482782363892, "learning_rate": 1.5451174289245984e-05, "epoch": 23.659139456820263, "step": 19400}, {"loss": 1.3942, "grad_norm": 0.5878770351409912, "learning_rate": 1.542645241038319e-05, "epoch": 23.781202319194385, "step": 19500}, {"loss": 1.3948, "grad_norm": 0.5305746793746948, "learning_rate": 1.5401730531520395e-05, "epoch": 23.903265181568507, "step": 19600}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 23.0, "global_step": 18860, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.064570709166719e+19, "log_history": [{"loss": 1.4006, "grad_norm": 0.5801668763160706, "learning_rate": 1.57725587144623e-05, "epoch": 22.073237717424472, "step": 18100}, {"loss": 1.3864, "grad_norm": 0.6067273616790771, "learning_rate": 1.5747836835599507e-05, "epoch": 22.195300579798598, "step": 18200}, {"loss": 1.3887, "grad_norm": 0.5616521239280701, "learning_rate": 1.5723114956736714e-05, "epoch": 22.31736344217272, "step": 18300}, {"loss": 1.4013, "grad_norm": 0.5156942009925842, "learning_rate": 1.5698393077873918e-05, "epoch": 22.439426304546842, "step": 18400}, {"loss": 1.4037, "grad_norm": 0.6042872071266174, "learning_rate": 1.5673671199011126e-05, "epoch": 22.561489166920964, "step": 18500}, {"loss": 1.4147, "grad_norm": 0.5273256897926331, "learning_rate": 1.5648949320148333e-05, "epoch": 22.683552029295086, "step": 18600}, {"loss": 1.389, "grad_norm": 0.6791280508041382, "learning_rate": 1.562422744128554e-05, "epoch": 22.805614891669208, "step": 18700}, {"loss": 1.3984, "grad_norm": 0.5689882040023804, "learning_rate": 1.5599505562422744e-05, "epoch": 22.927677754043334, "step": 18800}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 22.0, "global_step": 18040, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.019244793549907e+19, "log_history": [{"loss": 1.4181, "grad_norm": 0.5328198671340942, "learning_rate": 1.597033374536465e-05, "epoch": 21.0976502898993, "step": 17300}, {"loss": 1.4047, "grad_norm": 0.5192199945449829, "learning_rate": 1.5945611866501856e-05, "epoch": 21.21971315227342, "step": 17400}, {"loss": 1.4093, "grad_norm": 0.5063143968582153, "learning_rate": 1.592088998763906e-05, "epoch": 21.341776014647543, "step": 17500}, {"loss": 1.4032, "grad_norm": 0.47879233956336975, "learning_rate": 1.5896168108776267e-05, "epoch": 21.463838877021665, "step": 17600}, {"loss": 1.402, "grad_norm": 0.5259631872177124, "learning_rate": 1.5871446229913474e-05, "epoch": 21.585901739395787, "step": 17700}, {"loss": 1.3956, "grad_norm": 0.5722903609275818, "learning_rate": 1.584672435105068e-05, "epoch": 21.707964601769913, "step": 17800}, {"loss": 1.4044, "grad_norm": 0.5338879823684692, "learning_rate": 1.582200247218789e-05, "epoch": 21.830027464144035, "step": 17900}, {"loss": 1.4042, "grad_norm": 0.5523388981819153, "learning_rate": 1.5797280593325096e-05, "epoch": 21.952090326518157, "step": 18000}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 21.0, "global_step": 17220, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 9.739243512761518e+18, "log_history": [{"loss": 1.4067, "grad_norm": 0.5696863532066345, "learning_rate": 1.6168108776266997e-05, "epoch": 20.122062862374122, "step": 16500}, {"loss": 1.4122, "grad_norm": 0.37389010190963745, "learning_rate": 1.61433868974042e-05, "epoch": 20.244125724748244, "step": 16600}, {"loss": 1.4152, "grad_norm": 0.5331494808197021, "learning_rate": 1.6118665018541412e-05, "epoch": 20.36618858712237, "step": 16700}, {"loss": 1.4119, "grad_norm": 0.49912646412849426, "learning_rate": 1.6093943139678616e-05, "epoch": 20.48825144949649, "step": 16800}, {"loss": 1.4114, "grad_norm": 0.4720075726509094, "learning_rate": 1.6069221260815823e-05, "epoch": 20.610314311870614, "step": 16900}, {"loss": 1.4067, "grad_norm": 0.526085376739502, "learning_rate": 1.604449938195303e-05, "epoch": 20.732377174244736, "step": 17000}, {"loss": 1.4085, "grad_norm": 0.512604832649231, "learning_rate": 1.6019777503090238e-05, "epoch": 20.854440036618858, "step": 17100}, {"loss": 1.402, "grad_norm": 0.5988802313804626, "learning_rate": 1.599505562422744e-05, "epoch": 20.976502898992983, "step": 17200}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 20.0, "global_step": 16400, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 9.284961082349906e+18, "log_history": [{"loss": 1.4223, "grad_norm": 0.43760162591934204, "learning_rate": 1.639060568603214e-05, "epoch": 19.024412572474823, "step": 15600}, {"loss": 1.4088, "grad_norm": 0.5141226053237915, "learning_rate": 1.6365883807169346e-05, "epoch": 19.14647543484895, "step": 15700}, {"loss": 1.4004, "grad_norm": 0.47978028655052185, "learning_rate": 1.6341161928306553e-05, "epoch": 19.26853829722307, "step": 15800}, {"loss": 1.4241, "grad_norm": 0.49849799275398254, "learning_rate": 1.6316440049443757e-05, "epoch": 19.390601159597193, "step": 15900}, {"loss": 1.4314, "grad_norm": 0.44313570857048035, "learning_rate": 1.6291718170580965e-05, "epoch": 19.512664021971315, "step": 16000}, {"loss": 1.4218, "grad_norm": 0.6235785484313965, "learning_rate": 1.6266996291718172e-05, "epoch": 19.634726884345437, "step": 16100}, {"loss": 1.4143, "grad_norm": 0.523816704750061, "learning_rate": 1.624227441285538e-05, "epoch": 19.756789746719562, "step": 16200}, {"loss": 1.4128, "grad_norm": 0.5632780194282532, "learning_rate": 1.6217552533992583e-05, "epoch": 19.878852609093684, "step": 16300}, {"loss": 1.4157, "grad_norm": 5.377627372741699, "learning_rate": 1.6192830655129794e-05, "epoch": 20.0, "step": 16400}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 19.0, "global_step": 15580, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 8.776769464956211e+18, "log_history": [{"loss": 1.4306, "grad_norm": 0.4977523684501648, "learning_rate": 1.6588380716934488e-05, "epoch": 18.04882514494965, "step": 14800}, {"loss": 1.4144, "grad_norm": 0.5644989013671875, "learning_rate": 1.6563658838071695e-05, "epoch": 18.17088800732377, "step": 14900}, {"loss": 1.4416, "grad_norm": 0.47940465807914734, "learning_rate": 1.6538936959208902e-05, "epoch": 18.292950869697894, "step": 15000}, {"loss": 1.4078, "grad_norm": 0.5563494563102722, "learning_rate": 1.6514215080346106e-05, "epoch": 18.415013732072016, "step": 15100}, {"loss": 1.4188, "grad_norm": 0.4471091628074646, "learning_rate": 1.6489493201483313e-05, "epoch": 18.53707659444614, "step": 15200}, {"loss": 1.4282, "grad_norm": 0.5421698689460754, "learning_rate": 1.646477132262052e-05, "epoch": 18.659139456820263, "step": 15300}, {"loss": 1.4223, "grad_norm": 0.5628710389137268, "learning_rate": 1.6440049443757728e-05, "epoch": 18.781202319194385, "step": 15400}, {"loss": 1.4109, "grad_norm": 0.5217874050140381, "learning_rate": 1.6415327564894935e-05, "epoch": 18.903265181568507, "step": 15500}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 18.0, "global_step": 14760, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 8.323819211571886e+18, "log_history": [{"loss": 1.4144, "grad_norm": 0.48193618655204773, "learning_rate": 1.6786155747836836e-05, "epoch": 17.073237717424472, "step": 14000}, {"loss": 1.4261, "grad_norm": 0.4241122603416443, "learning_rate": 1.6761433868974044e-05, "epoch": 17.195300579798598, "step": 14100}, {"loss": 1.4261, "grad_norm": 0.45802411437034607, "learning_rate": 1.673671199011125e-05, "epoch": 17.31736344217272, "step": 14200}, {"loss": 1.4442, "grad_norm": 0.41032370924949646, "learning_rate": 1.6711990111248458e-05, "epoch": 17.439426304546842, "step": 14300}, {"loss": 1.4321, "grad_norm": 1.0284900665283203, "learning_rate": 1.6687268232385662e-05, "epoch": 17.561489166920964, "step": 14400}, {"loss": 1.4282, "grad_norm": 0.49651291966438293, "learning_rate": 1.666254635352287e-05, "epoch": 17.683552029295086, "step": 14500}, {"loss": 1.4366, "grad_norm": 0.5129456520080566, "learning_rate": 1.6637824474660076e-05, "epoch": 17.805614891669208, "step": 14600}, {"loss": 1.4195, "grad_norm": 0.44707387685775757, "learning_rate": 1.661310259579728e-05, "epoch": 17.927677754043334, "step": 14700}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 17.0, "global_step": 13940, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 7.871958831150766e+18, "log_history": [{"loss": 1.4416, "grad_norm": 0.44409170746803284, "learning_rate": 1.6983930778739185e-05, "epoch": 16.0976502898993, "step": 13200}, {"loss": 1.4244, "grad_norm": 0.44295036792755127, "learning_rate": 1.6959208899876392e-05, "epoch": 16.21971315227342, "step": 13300}, {"loss": 1.4402, "grad_norm": 0.39797458052635193, "learning_rate": 1.69344870210136e-05, "epoch": 16.341776014647543, "step": 13400}, {"loss": 1.4456, "grad_norm": 0.5135968923568726, "learning_rate": 1.6909765142150803e-05, "epoch": 16.463838877021665, "step": 13500}, {"loss": 1.4299, "grad_norm": 0.4955993890762329, "learning_rate": 1.688504326328801e-05, "epoch": 16.585901739395787, "step": 13600}, {"loss": 1.4379, "grad_norm": 0.5225348472595215, "learning_rate": 1.6860321384425218e-05, "epoch": 16.707964601769913, "step": 13700}, {"loss": 1.4298, "grad_norm": 1.2143789529800415, "learning_rate": 1.6835599505562422e-05, "epoch": 16.830027464144035, "step": 13800}, {"loss": 1.4176, "grad_norm": 0.41880977153778076, "learning_rate": 1.6810877626699632e-05, "epoch": 16.952090326518157, "step": 13900}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 16.0, "global_step": 13120, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 7.418293958644163e+18, "log_history": [{"loss": 1.4426, "grad_norm": 0.42590752243995667, "learning_rate": 1.7181705809641534e-05, "epoch": 15.122062862374122, "step": 12400}, {"loss": 1.4284, "grad_norm": 0.41311484575271606, "learning_rate": 1.715698393077874e-05, "epoch": 15.244125724748246, "step": 12500}, {"loss": 1.4473, "grad_norm": 0.6427305340766907, "learning_rate": 1.7132262051915945e-05, "epoch": 15.366188587122368, "step": 12600}, {"loss": 1.4302, "grad_norm": 0.4556065797805786, "learning_rate": 1.7107540173053156e-05, "epoch": 15.488251449496492, "step": 12700}, {"loss": 1.4365, "grad_norm": 0.3812936544418335, "learning_rate": 1.708281829419036e-05, "epoch": 15.610314311870614, "step": 12800}, {"loss": 1.4481, "grad_norm": 0.578228235244751, "learning_rate": 1.7058096415327567e-05, "epoch": 15.732377174244736, "step": 12900}, {"loss": 1.4278, "grad_norm": 0.527608335018158, "learning_rate": 1.7033374536464774e-05, "epoch": 15.85444003661886, "step": 13000}, {"loss": 1.4491, "grad_norm": 0.43727609515190125, "learning_rate": 1.7008652657601978e-05, "epoch": 15.976502898992981, "step": 13100}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 15.0, "global_step": 12300, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 6.963969020327608e+18, "log_history": [{"loss": 1.4545, "grad_norm": 0.4767194092273712, "learning_rate": 1.7404202719406675e-05, "epoch": 14.024412572474825, "step": 11500}, {"loss": 1.4461, "grad_norm": 0.3866611123085022, "learning_rate": 1.7379480840543883e-05, "epoch": 14.146475434848947, "step": 11600}, {"loss": 1.448, "grad_norm": 0.3985901176929474, "learning_rate": 1.735475896168109e-05, "epoch": 14.26853829722307, "step": 11700}, {"loss": 1.4415, "grad_norm": 0.4791412055492401, "learning_rate": 1.7330037082818297e-05, "epoch": 14.390601159597193, "step": 11800}, {"loss": 1.4458, "grad_norm": 0.4854709506034851, "learning_rate": 1.73053152039555e-05, "epoch": 14.512664021971315, "step": 11900}, {"loss": 1.4363, "grad_norm": 0.38165754079818726, "learning_rate": 1.7280593325092708e-05, "epoch": 14.634726884345438, "step": 12000}, {"loss": 1.4415, "grad_norm": 0.4177747368812561, "learning_rate": 1.7255871446229915e-05, "epoch": 14.75678974671956, "step": 12100}, {"loss": 1.4394, "grad_norm": 0.47537434101104736, "learning_rate": 1.723114956736712e-05, "epoch": 14.878852609093684, "step": 12200}, {"loss": 1.4707, "grad_norm": 1.2642663717269897, "learning_rate": 1.7206427688504327e-05, "epoch": 15.0, "step": 12300}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 14.0, "global_step": 11480, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 6.454806813275996e+18, "log_history": [{"loss": 1.4557, "grad_norm": 0.38796746730804443, "learning_rate": 1.7601977750309024e-05, "epoch": 13.04882514494965, "step": 10700}, {"loss": 1.4619, "grad_norm": 0.37647825479507446, "learning_rate": 1.757725587144623e-05, "epoch": 13.170888007323772, "step": 10800}, {"loss": 1.4414, "grad_norm": 0.431555837392807, "learning_rate": 1.755253399258344e-05, "epoch": 13.292950869697894, "step": 10900}, {"loss": 1.4524, "grad_norm": 0.40302348136901855, "learning_rate": 1.7527812113720642e-05, "epoch": 13.415013732072017, "step": 11000}, {"loss": 1.4499, "grad_norm": 0.4036793112754822, "learning_rate": 1.750309023485785e-05, "epoch": 13.53707659444614, "step": 11100}, {"loss": 1.4551, "grad_norm": 0.39906376600265503, "learning_rate": 1.7478368355995057e-05, "epoch": 13.659139456820263, "step": 11200}, {"loss": 1.4672, "grad_norm": 0.41409650444984436, "learning_rate": 1.7453646477132264e-05, "epoch": 13.781202319194385, "step": 11300}, {"loss": 1.4498, "grad_norm": 0.43862468004226685, "learning_rate": 1.742892459826947e-05, "epoch": 13.903265181568507, "step": 11400}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 13.0, "global_step": 10660, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 6.002558233015357e+18, "log_history": [{"loss": 1.4563, "grad_norm": 0.39366087317466736, "learning_rate": 1.7799752781211376e-05, "epoch": 12.073237717424474, "step": 9900}, {"loss": 1.4619, "grad_norm": 0.4358100891113281, "learning_rate": 1.777503090234858e-05, "epoch": 12.195300579798596, "step": 10000}, {"loss": 1.4538, "grad_norm": 0.415566086769104, "learning_rate": 1.7750309023485784e-05, "epoch": 12.317363442172718, "step": 10100}, {"loss": 1.4444, "grad_norm": 0.42875242233276367, "learning_rate": 1.7725587144622995e-05, "epoch": 12.439426304546842, "step": 10200}, {"loss": 1.4615, "grad_norm": 0.43399542570114136, "learning_rate": 1.77008652657602e-05, "epoch": 12.561489166920964, "step": 10300}, {"loss": 1.4634, "grad_norm": 0.3900112211704254, "learning_rate": 1.7676143386897406e-05, "epoch": 12.683552029295086, "step": 10400}, {"loss": 1.4692, "grad_norm": 0.43448755145072937, "learning_rate": 1.7651421508034613e-05, "epoch": 12.80561489166921, "step": 10500}, {"loss": 1.4667, "grad_norm": 0.3975508213043213, "learning_rate": 1.762669962917182e-05, "epoch": 12.927677754043332, "step": 10600}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 12.0, "global_step": 9840, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 5.550099297163684e+18, "log_history": [{"loss": 1.4795, "grad_norm": 0.3457795977592468, "learning_rate": 1.799752781211372e-05, "epoch": 11.097650289899299, "step": 9100}, {"loss": 1.451, "grad_norm": 0.41427910327911377, "learning_rate": 1.797280593325093e-05, "epoch": 11.219713152273421, "step": 9200}, {"loss": 1.4614, "grad_norm": 0.3713468015193939, "learning_rate": 1.7948084054388136e-05, "epoch": 11.341776014647543, "step": 9300}, {"loss": 1.4734, "grad_norm": 0.3347463309764862, "learning_rate": 1.792336217552534e-05, "epoch": 11.463838877021667, "step": 9400}, {"loss": 1.4583, "grad_norm": 0.3549976348876953, "learning_rate": 1.7898640296662547e-05, "epoch": 11.585901739395789, "step": 9500}, {"loss": 1.4591, "grad_norm": 0.4143538475036621, "learning_rate": 1.7873918417799754e-05, "epoch": 11.70796460176991, "step": 9600}, {"loss": 1.4684, "grad_norm": 0.43329042196273804, "learning_rate": 1.784919653893696e-05, "epoch": 11.830027464144035, "step": 9700}, {"loss": 1.486, "grad_norm": 0.38120603561401367, "learning_rate": 1.7824474660074166e-05, "epoch": 11.952090326518157, "step": 9800}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 11.0, "global_step": 9020, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 5.097770992066499e+18, "log_history": [{"loss": 1.4614, "grad_norm": 0.37912508845329285, "learning_rate": 1.819530284301607e-05, "epoch": 10.122062862374122, "step": 8300}, {"loss": 1.4683, "grad_norm": 0.3781580328941345, "learning_rate": 1.8170580964153278e-05, "epoch": 10.244125724748246, "step": 8400}, {"loss": 1.4708, "grad_norm": 0.3734969198703766, "learning_rate": 1.814585908529048e-05, "epoch": 10.366188587122368, "step": 8500}, {"loss": 1.4636, "grad_norm": 0.4024413228034973, "learning_rate": 1.812113720642769e-05, "epoch": 10.488251449496492, "step": 8600}, {"loss": 1.4599, "grad_norm": 0.3641699552536011, "learning_rate": 1.8096415327564896e-05, "epoch": 10.610314311870614, "step": 8700}, {"loss": 1.4769, "grad_norm": 0.34929582476615906, "learning_rate": 1.8071693448702103e-05, "epoch": 10.732377174244736, "step": 8800}, {"loss": 1.4952, "grad_norm": 0.385298490524292, "learning_rate": 1.804697156983931e-05, "epoch": 10.85444003661886, "step": 8900}, {"loss": 1.4831, "grad_norm": 0.3819405436515808, "learning_rate": 1.8022249690976518e-05, "epoch": 10.976502898992981, "step": 9000}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 10.0, "global_step": 8200, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 4.643918413837701e+18, "log_history": [{"loss": 1.4847, "grad_norm": 0.380543053150177, "learning_rate": 1.8417799752781215e-05, "epoch": 9.024412572474825, "step": 7400}, {"loss": 1.4676, "grad_norm": 0.33778199553489685, "learning_rate": 1.839307787391842e-05, "epoch": 9.146475434848947, "step": 7500}, {"loss": 1.4745, "grad_norm": 0.4859809875488281, "learning_rate": 1.8368355995055626e-05, "epoch": 9.26853829722307, "step": 7600}, {"loss": 1.4705, "grad_norm": 0.39950817823410034, "learning_rate": 1.8343634116192834e-05, "epoch": 9.390601159597193, "step": 7700}, {"loss": 1.4741, "grad_norm": 0.3742108643054962, "learning_rate": 1.8318912237330037e-05, "epoch": 9.512664021971315, "step": 7800}, {"loss": 1.4862, "grad_norm": 0.36551040410995483, "learning_rate": 1.8294190358467245e-05, "epoch": 9.634726884345438, "step": 7900}, {"loss": 1.479, "grad_norm": 0.44601181149482727, "learning_rate": 1.8269468479604452e-05, "epoch": 9.75678974671956, "step": 8000}, {"loss": 1.4903, "grad_norm": 0.3847127854824066, "learning_rate": 1.824474660074166e-05, "epoch": 9.878852609093684, "step": 8100}, {"loss": 1.4868, "grad_norm": 1.8134512901306152, "learning_rate": 1.8220024721878863e-05, "epoch": 10.0, "step": 8200}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 9.0, "global_step": 7380, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 4.133047650178867e+18, "log_history": [{"loss": 1.4787, "grad_norm": 0.4521217942237854, "learning_rate": 1.861557478368356e-05, "epoch": 8.04882514494965, "step": 6600}, {"loss": 1.4939, "grad_norm": 0.3527626097202301, "learning_rate": 1.8590852904820768e-05, "epoch": 8.170888007323772, "step": 6700}, {"loss": 1.4958, "grad_norm": 0.36873555183410645, "learning_rate": 1.8566131025957975e-05, "epoch": 8.292950869697894, "step": 6800}, {"loss": 1.4757, "grad_norm": 0.3399360477924347, "learning_rate": 1.8541409147095182e-05, "epoch": 8.415013732072017, "step": 6900}, {"loss": 1.4641, "grad_norm": 0.4521547853946686, "learning_rate": 1.8516687268232386e-05, "epoch": 8.53707659444614, "step": 7000}, {"loss": 1.4974, "grad_norm": 0.34370288252830505, "learning_rate": 1.8491965389369593e-05, "epoch": 8.659139456820263, "step": 7100}, {"loss": 1.4958, "grad_norm": 0.44971323013305664, "learning_rate": 1.84672435105068e-05, "epoch": 8.781202319194385, "step": 7200}, {"loss": 1.5089, "grad_norm": 0.5601497888565063, "learning_rate": 1.8442521631644004e-05, "epoch": 8.903265181568507, "step": 7300}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 8.0, "global_step": 6560, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 3.682653342182646e+18, "log_history": [{"loss": 1.4929, "grad_norm": 0.33132004737854004, "learning_rate": 1.881334981458591e-05, "epoch": 7.073237717424473, "step": 5800}, {"loss": 1.4977, "grad_norm": 0.3214746415615082, "learning_rate": 1.8788627935723116e-05, "epoch": 7.195300579798596, "step": 5900}, {"loss": 1.5028, "grad_norm": 0.3732578754425049, "learning_rate": 1.8763906056860324e-05, "epoch": 7.317363442172719, "step": 6000}, {"loss": 1.48, "grad_norm": 0.365268737077713, "learning_rate": 1.8739184177997528e-05, "epoch": 7.439426304546842, "step": 6100}, {"loss": 1.4969, "grad_norm": 0.3413032293319702, "learning_rate": 1.8714462299134735e-05, "epoch": 7.561489166920964, "step": 6200}, {"loss": 1.4963, "grad_norm": 0.37481433153152466, "learning_rate": 1.8689740420271942e-05, "epoch": 7.683552029295087, "step": 6300}, {"loss": 1.4958, "grad_norm": 0.37407252192497253, "learning_rate": 1.866501854140915e-05, "epoch": 7.80561489166921, "step": 6400}, {"loss": 1.5079, "grad_norm": 0.38370847702026367, "learning_rate": 1.8640296662546357e-05, "epoch": 7.927677754043332, "step": 6500}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 7.0, "global_step": 5740, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 3.227777444680796e+18, "log_history": [{"loss": 1.5028, "grad_norm": 0.2831012010574341, "learning_rate": 1.9011124845488258e-05, "epoch": 6.097650289899298, "step": 5000}, {"loss": 1.5147, "grad_norm": 0.33215996623039246, "learning_rate": 1.8986402966625465e-05, "epoch": 6.219713152273421, "step": 5100}, {"loss": 1.5207, "grad_norm": 0.27129489183425903, "learning_rate": 1.8961681087762672e-05, "epoch": 6.341776014647543, "step": 5200}, {"loss": 1.5083, "grad_norm": 0.3230811357498169, "learning_rate": 1.893695920889988e-05, "epoch": 6.463838877021666, "step": 5300}, {"loss": 1.4922, "grad_norm": 0.34408777952194214, "learning_rate": 1.8912237330037084e-05, "epoch": 6.585901739395789, "step": 5400}, {"loss": 1.5122, "grad_norm": 0.30923399329185486, "learning_rate": 1.888751545117429e-05, "epoch": 6.707964601769912, "step": 5500}, {"loss": 1.5085, "grad_norm": 0.2929703891277313, "learning_rate": 1.8862793572311498e-05, "epoch": 6.830027464144035, "step": 5600}, {"loss": 1.4865, "grad_norm": 0.3233591914176941, "learning_rate": 1.8838071693448702e-05, "epoch": 6.952090326518157, "step": 5700}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 6.0, "global_step": 4920, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 2.775053464837079e+18, "log_history": [{"loss": 1.5189, "grad_norm": 0.30038759112358093, "learning_rate": 1.9208899876390607e-05, "epoch": 5.122062862374123, "step": 4200}, {"loss": 1.5399, "grad_norm": 0.3732454776763916, "learning_rate": 1.9184177997527814e-05, "epoch": 5.244125724748246, "step": 4300}, {"loss": 1.5037, "grad_norm": 0.3264354467391968, "learning_rate": 1.915945611866502e-05, "epoch": 5.366188587122368, "step": 4400}, {"loss": 1.5007, "grad_norm": 0.3741842806339264, "learning_rate": 1.9134734239802225e-05, "epoch": 5.488251449496491, "step": 4500}, {"loss": 1.4941, "grad_norm": 0.30297067761421204, "learning_rate": 1.9110012360939432e-05, "epoch": 5.610314311870614, "step": 4600}, {"loss": 1.5181, "grad_norm": 0.3293197751045227, "learning_rate": 1.908529048207664e-05, "epoch": 5.732377174244736, "step": 4700}, {"loss": 1.5319, "grad_norm": 0.37035322189331055, "learning_rate": 1.9060568603213843e-05, "epoch": 5.8544400366188585, "step": 4800}, {"loss": 1.514, "grad_norm": 0.31698599457740784, "learning_rate": 1.9035846724351054e-05, "epoch": 5.976502898992981, "step": 4900}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 5.0, "global_step": 4100, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 2.3214545831511245e+18, "log_history": [{"loss": 1.5351, "grad_norm": 0.3526078760623932, "learning_rate": 1.9431396786155748e-05, "epoch": 4.024412572474825, "step": 3300}, {"loss": 1.5243, "grad_norm": 0.30970993638038635, "learning_rate": 1.9406674907292955e-05, "epoch": 4.146475434848947, "step": 3400}, {"loss": 1.5227, "grad_norm": 0.2702256143093109, "learning_rate": 1.9381953028430163e-05, "epoch": 4.26853829722307, "step": 3500}, {"loss": 1.534, "grad_norm": 0.2982151508331299, "learning_rate": 1.9357231149567367e-05, "epoch": 4.3906011595971925, "step": 3600}, {"loss": 1.5251, "grad_norm": 0.285404235124588, "learning_rate": 1.9332509270704577e-05, "epoch": 4.512664021971315, "step": 3700}, {"loss": 1.5392, "grad_norm": 0.33566904067993164, "learning_rate": 1.930778739184178e-05, "epoch": 4.634726884345438, "step": 3800}, {"loss": 1.5243, "grad_norm": 0.2565667927265167, "learning_rate": 1.928306551297899e-05, "epoch": 4.75678974671956, "step": 3900}, {"loss": 1.5205, "grad_norm": 0.2805371880531311, "learning_rate": 1.9258343634116196e-05, "epoch": 4.878852609093683, "step": 4000}, {"loss": 1.5233, "grad_norm": 0.8891302943229675, "learning_rate": 1.92336217552534e-05, "epoch": 5.0, "step": 4100}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 4.0, "global_step": 3280, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.8119575137741926e+18, "log_history": [{"loss": 1.5434, "grad_norm": 0.21596193313598633, "learning_rate": 1.9629171817058097e-05, "epoch": 3.048825144949649, "step": 2500}, {"loss": 1.5479, "grad_norm": 0.2698516249656677, "learning_rate": 1.9604449938195304e-05, "epoch": 3.1708880073237715, "step": 2600}, {"loss": 1.5318, "grad_norm": 0.25629115104675293, "learning_rate": 1.957972805933251e-05, "epoch": 3.2929508696978944, "step": 2700}, {"loss": 1.5458, "grad_norm": 0.24703611433506012, "learning_rate": 1.955500618046972e-05, "epoch": 3.4150137320720173, "step": 2800}, {"loss": 1.532, "grad_norm": 0.7753630876541138, "learning_rate": 1.9530284301606923e-05, "epoch": 3.5370765944461398, "step": 2900}, {"loss": 1.548, "grad_norm": 0.2873762547969818, "learning_rate": 1.950556242274413e-05, "epoch": 3.659139456820262, "step": 3000}, {"loss": 1.529, "grad_norm": 0.24021025002002716, "learning_rate": 1.9480840543881337e-05, "epoch": 3.781202319194385, "step": 3100}, {"loss": 1.5542, "grad_norm": 0.3195001780986786, "learning_rate": 1.9456118665018544e-05, "epoch": 3.903265181568508, "step": 3200}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 3.0, "global_step": 2460, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 1.3586553093464064e+18, "log_history": [{"loss": 1.5697, "grad_norm": 0.19002500176429749, "learning_rate": 1.9826946847960446e-05, "epoch": 2.0732377174244734, "step": 1700}, {"loss": 1.5758, "grad_norm": 0.19588354229927063, "learning_rate": 1.9802224969097653e-05, "epoch": 2.1953005797985963, "step": 1800}, {"loss": 1.5639, "grad_norm": 0.25814926624298096, "learning_rate": 1.977750309023486e-05, "epoch": 2.317363442172719, "step": 1900}, {"loss": 1.5709, "grad_norm": 0.40624576807022095, "learning_rate": 1.9752781211372064e-05, "epoch": 2.4394263045468416, "step": 2000}, {"loss": 1.5722, "grad_norm": 0.2429089993238449, "learning_rate": 1.972805933250927e-05, "epoch": 2.561489166920964, "step": 2100}, {"loss": 1.5526, "grad_norm": 0.3574996590614319, "learning_rate": 1.970333745364648e-05, "epoch": 2.683552029295087, "step": 2200}, {"loss": 1.5615, "grad_norm": 0.20724007487297058, "learning_rate": 1.9678615574783686e-05, "epoch": 2.80561489166921, "step": 2300}, {"loss": 1.5436, "grad_norm": 0.24733370542526245, "learning_rate": 1.9653893695920893e-05, "epoch": 2.9276777540433323, "step": 2400}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 2.0, "global_step": 1640, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 9.067051174672282e+17, "log_history": [{"loss": 1.6331, "grad_norm": 0.1402634233236313, "learning_rate": 1.8e-05, "epoch": 1.0976502898992981, "step": 900}, {"loss": 1.6271, "grad_norm": 0.4210573732852936, "learning_rate": 2e-05, "epoch": 1.2197131522734208, "step": 1000}, {"loss": 1.6148, "grad_norm": 0.14865167438983917, "learning_rate": 1.9975278121137206e-05, "epoch": 1.3417760146475435, "step": 1100}, {"loss": 1.6013, "grad_norm": 0.3981769382953644, "learning_rate": 1.9950556242274416e-05, "epoch": 1.4638388770216662, "step": 1200}, {"loss": 1.6032, "grad_norm": 0.23847170174121857, "learning_rate": 1.992583436341162e-05, "epoch": 1.5859017393957888, "step": 1300}, {"loss": 1.5971, "grad_norm": 0.16641396284103394, "learning_rate": 1.9901112484548827e-05, "epoch": 1.7079646017699115, "step": 1400}, {"loss": 1.5604, "grad_norm": 0.17072312533855438, "learning_rate": 1.9876390605686035e-05, "epoch": 1.8300274641440342, "step": 1500}, {"loss": 1.5755, "grad_norm": 0.17055442929267883, "learning_rate": 1.9851668726823242e-05, "epoch": 1.9520903265181568, "step": 1600}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 1.0, "global_step": 820, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 4.5308014115106816e+17, "log_history": [{"loss": 1.9953, "grad_norm": 0.1531936526298523, "learning_rate": 2.0000000000000003e-06, "epoch": 0.12206286237412267, "step": 100}, {"loss": 1.9693, "grad_norm": 0.1935437172651291, "learning_rate": 4.000000000000001e-06, "epoch": 0.24412572474824534, "step": 200}, {"loss": 1.8897, "grad_norm": 0.15954454243183136, "learning_rate": 6e-06, "epoch": 0.366188587122368, "step": 300}, {"loss": 1.7581, "grad_norm": 0.13417589664459229, "learning_rate": 8.000000000000001e-06, "epoch": 0.4882514494964907, "step": 400}, {"loss": 1.723, "grad_norm": 0.09929320961236954, "learning_rate": 1e-05, "epoch": 0.6103143118706134, "step": 500}, {"loss": 1.6892, "grad_norm": 0.0994492769241333, "learning_rate": 1.2e-05, "epoch": 0.732377174244736, "step": 600}, {"loss": 1.666, "grad_norm": 0.12159933149814606, "learning_rate": 1.4e-05, "epoch": 0.8544400366188587, "step": 700}, {"loss": 1.6738, "grad_norm": 0.17567557096481323, "learning_rate": 1.6000000000000003e-05, "epoch": 0.9765028989929814, "step": 800}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 0, "global_step": 0, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 0, "log_history": [], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": false, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 0, "global_step": 0, "max_steps": 81900, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 4, "num_train_epochs": 100, "num_input_tokens_seen": 0, "total_flos": 0, "log_history": [], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": false, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
Upload README.md with huggingface_hub
initial commit
