avsolatorio/data-use-unsloth-phi-3.5-data-newschema-tccml-iclr-prwp-train-50epochs-1738184999-lora
{"epoch": 10.0, "global_step": 13870, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 3.222952234600366e+18, "log_history": [{"loss": 0.8267, "grad_norm": 0.5286613702774048, "learning_rate": 0.0004158125915080527, "epoch": 9.012263300270513, "step": 12500}, {"loss": 0.621, "grad_norm": 0.5431559681892395, "learning_rate": 0.0004150805270863836, "epoch": 9.084400360685303, "step": 12600}, {"loss": 0.6404, "grad_norm": 0.6276618838310242, "learning_rate": 0.00041434846266471447, "epoch": 9.15653742110009, "step": 12700}, {"loss": 0.6834, "grad_norm": 0.6356784701347351, "learning_rate": 0.0004136163982430454, "epoch": 9.228674481514878, "step": 12800}, {"loss": 0.6934, "grad_norm": 0.6668227910995483, "learning_rate": 0.0004128843338213763, "epoch": 9.300811541929667, "step": 12900}, {"loss": 0.7205, "grad_norm": 0.7019756436347961, "learning_rate": 0.0004121522693997072, "epoch": 9.372948602344454, "step": 13000}, {"loss": 0.7221, "grad_norm": 0.7371429800987244, "learning_rate": 0.0004114202049780381, "epoch": 9.445085662759242, "step": 13100}, {"loss": 0.7444, "grad_norm": 0.6809977889060974, "learning_rate": 0.000410688140556369, "epoch": 9.517222723174031, "step": 13200}, {"loss": 0.7574, "grad_norm": 0.6309143900871277, "learning_rate": 0.0004099560761346999, "epoch": 9.589359783588819, "step": 13300}, {"loss": 0.7733, "grad_norm": 0.6243112087249756, "learning_rate": 0.00040922401171303074, "epoch": 9.661496844003606, "step": 13400}, {"loss": 0.7718, "grad_norm": 0.7073637247085571, "learning_rate": 0.00040849194729136165, "epoch": 9.733633904418395, "step": 13500}, {"loss": 0.7741, "grad_norm": 0.6300414800643921, "learning_rate": 0.00040775988286969255, "epoch": 9.805770964833183, "step": 13600}, {"loss": 0.7868, "grad_norm": 0.6416450142860413, "learning_rate": 0.00040702781844802346, "epoch": 9.87790802524797, "step": 13700}, {"loss": 0.8136, "grad_norm": 0.6355851888656616, "learning_rate": 0.0004062957540263543, "epoch": 9.95004508566276, "step": 13800}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 9.0, "global_step": 12483, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 2.895835804187812e+18, "log_history": [{"loss": 0.934, "grad_norm": 0.43410107493400574, "learning_rate": 0.0004260614934114202, "epoch": 8.002885482416591, "step": 11100}, {"loss": 0.685, "grad_norm": 0.6895626783370972, "learning_rate": 0.0004253294289897511, "epoch": 8.07502254283138, "step": 11200}, {"loss": 0.7173, "grad_norm": 0.6718540787696838, "learning_rate": 0.000424597364568082, "epoch": 8.147159603246168, "step": 11300}, {"loss": 0.7462, "grad_norm": 0.6727272868156433, "learning_rate": 0.00042386530014641287, "epoch": 8.219296663660955, "step": 11400}, {"loss": 0.7619, "grad_norm": 0.651489794254303, "learning_rate": 0.0004231332357247438, "epoch": 8.291433724075745, "step": 11500}, {"loss": 0.7718, "grad_norm": 0.5339372754096985, "learning_rate": 0.0004224011713030747, "epoch": 8.363570784490532, "step": 11600}, {"loss": 0.8075, "grad_norm": 0.7253825664520264, "learning_rate": 0.0004216691068814056, "epoch": 8.43570784490532, "step": 11700}, {"loss": 0.8361, "grad_norm": 1.0222340822219849, "learning_rate": 0.0004209370424597365, "epoch": 8.507844905320109, "step": 11800}, {"loss": 0.8272, "grad_norm": 0.9203081727027893, "learning_rate": 0.00042020497803806734, "epoch": 8.579981965734897, "step": 11900}, {"loss": 0.8351, "grad_norm": 1.2445309162139893, "learning_rate": 0.0004194729136163983, "epoch": 8.652119026149684, "step": 12000}, {"loss": 0.8692, "grad_norm": 0.6743190288543701, "learning_rate": 0.00041874084919472915, "epoch": 8.724256086564473, "step": 12100}, {"loss": 0.8669, "grad_norm": 0.6785947680473328, "learning_rate": 0.00041800878477306005, "epoch": 8.796393146979261, "step": 12200}, {"loss": 0.8596, "grad_norm": 0.7174391746520996, "learning_rate": 0.0004172767203513909, "epoch": 8.868530207394048, "step": 12300}, {"loss": 0.8719, "grad_norm": 0.6030064821243286, "learning_rate": 0.00041654465592972186, "epoch": 8.940667267808838, "step": 12400}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 8.0, "global_step": 11096, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 2.567483537492275e+18, "log_history": [{"loss": 0.8165, "grad_norm": 0.5385152697563171, "learning_rate": 0.0004355783308931186, "epoch": 7.065644724977457, "step": 9800}, {"loss": 0.7963, "grad_norm": 0.45924773812294006, "learning_rate": 0.00043484626647144947, "epoch": 7.137781785392245, "step": 9900}, {"loss": 0.8437, "grad_norm": 0.6439551711082458, "learning_rate": 0.00043411420204978043, "epoch": 7.209918845807033, "step": 10000}, {"loss": 0.8487, "grad_norm": 0.5916787981987, "learning_rate": 0.0004333821376281113, "epoch": 7.282055906221822, "step": 10100}, {"loss": 0.8737, "grad_norm": 0.6494361758232117, "learning_rate": 0.0004326500732064422, "epoch": 7.354192966636609, "step": 10200}, {"loss": 0.8928, "grad_norm": 0.8666304349899292, "learning_rate": 0.00043191800878477303, "epoch": 7.426330027051398, "step": 10300}, {"loss": 0.8856, "grad_norm": 0.5604568719863892, "learning_rate": 0.000431185944363104, "epoch": 7.498467087466186, "step": 10400}, {"loss": 0.8941, "grad_norm": 0.5468480587005615, "learning_rate": 0.00043045387994143484, "epoch": 7.570604147880974, "step": 10500}, {"loss": 0.9342, "grad_norm": 0.5744903087615967, "learning_rate": 0.00042972181551976575, "epoch": 7.642741208295762, "step": 10600}, {"loss": 0.9242, "grad_norm": 0.5883183479309082, "learning_rate": 0.00042898975109809665, "epoch": 7.71487826871055, "step": 10700}, {"loss": 0.9519, "grad_norm": 0.5754871368408203, "learning_rate": 0.00042825768667642755, "epoch": 7.787015329125338, "step": 10800}, {"loss": 0.9501, "grad_norm": 0.5902408361434937, "learning_rate": 0.00042752562225475846, "epoch": 7.859152389540126, "step": 10900}, {"loss": 0.938, "grad_norm": 0.5574387907981873, "learning_rate": 0.0004267935578330893, "epoch": 7.931289449954915, "step": 11000}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 7.0, "global_step": 9709, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 2.2645854905953075e+18, "log_history": [{"loss": 0.9706, "grad_norm": 0.48168620467185974, "learning_rate": 0.0004458272327964861, "epoch": 6.056266907123534, "step": 8400}, {"loss": 0.8876, "grad_norm": 0.6274648904800415, "learning_rate": 0.000445095168374817, "epoch": 6.128403967538323, "step": 8500}, {"loss": 0.9196, "grad_norm": 0.5852373242378235, "learning_rate": 0.0004443631039531479, "epoch": 6.200541027953111, "step": 8600}, {"loss": 0.9444, "grad_norm": 0.4553835988044739, "learning_rate": 0.0004436310395314788, "epoch": 6.272678088367899, "step": 8700}, {"loss": 0.9659, "grad_norm": 0.49562057852745056, "learning_rate": 0.0004428989751098097, "epoch": 6.344815148782687, "step": 8800}, {"loss": 0.9794, "grad_norm": 0.5985817909240723, "learning_rate": 0.0004421669106881406, "epoch": 6.4169522091974756, "step": 8900}, {"loss": 0.98, "grad_norm": 0.46492019295692444, "learning_rate": 0.00044143484626647144, "epoch": 6.489089269612263, "step": 9000}, {"loss": 1.0016, "grad_norm": 0.4412674009799957, "learning_rate": 0.00044070278184480234, "epoch": 6.5612263300270515, "step": 9100}, {"loss": 0.9973, "grad_norm": 0.6077773571014404, "learning_rate": 0.00043997071742313325, "epoch": 6.63336339044184, "step": 9200}, {"loss": 1.0134, "grad_norm": 0.490397572517395, "learning_rate": 0.00043923865300146415, "epoch": 6.705500450856627, "step": 9300}, {"loss": 1.0172, "grad_norm": 0.4129747152328491, "learning_rate": 0.00043850658857979506, "epoch": 6.777637511271416, "step": 9400}, {"loss": 1.0352, "grad_norm": 0.4793046712875366, "learning_rate": 0.0004377745241581259, "epoch": 6.849774571686204, "step": 9500}, {"loss": 1.0377, "grad_norm": 0.4928447902202606, "learning_rate": 0.00043704245973645686, "epoch": 6.921911632100992, "step": 9600}, {"loss": 1.0526, "grad_norm": 0.5162463784217834, "learning_rate": 0.0004363103953147877, "epoch": 6.99404869251578, "step": 9700}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 6.0, "global_step": 8322, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 1.9384602058349568e+18, "log_history": [{"loss": 1.0875, "grad_norm": 0.5500540137290955, "learning_rate": 0.0004560761346998536, "epoch": 5.046889089269612, "step": 7000}, {"loss": 1.0078, "grad_norm": 0.4185388684272766, "learning_rate": 0.00045534407027818447, "epoch": 5.119026149684401, "step": 7100}, {"loss": 1.0239, "grad_norm": 0.39495837688446045, "learning_rate": 0.00045461200585651543, "epoch": 5.191163210099188, "step": 7200}, {"loss": 1.0217, "grad_norm": 0.37627801299095154, "learning_rate": 0.0004538799414348463, "epoch": 5.263300270513977, "step": 7300}, {"loss": 1.0612, "grad_norm": 0.48439398407936096, "learning_rate": 0.0004531478770131772, "epoch": 5.335437330928765, "step": 7400}, {"loss": 1.0852, "grad_norm": 0.4401755630970001, "learning_rate": 0.00045241581259150804, "epoch": 5.4075743913435526, "step": 7500}, {"loss": 1.0899, "grad_norm": 0.41516146063804626, "learning_rate": 0.000451683748169839, "epoch": 5.479711451758341, "step": 7600}, {"loss": 1.0898, "grad_norm": 0.37935078144073486, "learning_rate": 0.00045095168374816984, "epoch": 5.5518485121731285, "step": 7700}, {"loss": 1.1027, "grad_norm": 0.5189271569252014, "learning_rate": 0.00045021961932650075, "epoch": 5.623985572587917, "step": 7800}, {"loss": 1.1142, "grad_norm": 0.3512900471687317, "learning_rate": 0.0004494875549048316, "epoch": 5.696122633002705, "step": 7900}, {"loss": 1.1279, "grad_norm": 0.4437732398509979, "learning_rate": 0.00044875549048316256, "epoch": 5.768259693417493, "step": 8000}, {"loss": 1.1002, "grad_norm": 0.3989928066730499, "learning_rate": 0.0004480234260614934, "epoch": 5.840396753832281, "step": 8100}, {"loss": 1.1425, "grad_norm": 0.4574419856071472, "learning_rate": 0.0004472913616398243, "epoch": 5.91253381424707, "step": 8200}, {"loss": 1.152, "grad_norm": 0.4563077986240387, "learning_rate": 0.0004465592972181552, "epoch": 5.984670874661857, "step": 8300}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 5.0, "global_step": 6935, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 1.6111549664759808e+18, "log_history": [{"loss": 1.2451, "grad_norm": 0.4010903835296631, "learning_rate": 0.0004663250366032211, "epoch": 4.03751127141569, "step": 5600}, {"loss": 1.1346, "grad_norm": 0.37167391180992126, "learning_rate": 0.000465592972181552, "epoch": 4.109648331830478, "step": 5700}, {"loss": 1.1509, "grad_norm": 0.3437536060810089, "learning_rate": 0.0004648609077598829, "epoch": 4.181785392245266, "step": 5800}, {"loss": 1.157, "grad_norm": 0.3578796088695526, "learning_rate": 0.0004641288433382138, "epoch": 4.2539224526600545, "step": 5900}, {"loss": 1.1808, "grad_norm": 0.3500485122203827, "learning_rate": 0.0004633967789165447, "epoch": 4.326059513074842, "step": 6000}, {"loss": 1.2009, "grad_norm": 0.3375537693500519, "learning_rate": 0.0004626647144948756, "epoch": 4.3981965734896304, "step": 6100}, {"loss": 1.2071, "grad_norm": 0.4168888330459595, "learning_rate": 0.00046193265007320644, "epoch": 4.470333633904419, "step": 6200}, {"loss": 1.2133, "grad_norm": 0.36075296998023987, "learning_rate": 0.00046120058565153735, "epoch": 4.542470694319206, "step": 6300}, {"loss": 1.1537, "grad_norm": 0.36374935507774353, "learning_rate": 0.00046046852122986825, "epoch": 4.614607754733995, "step": 6400}, {"loss": 1.2067, "grad_norm": 0.42417722940444946, "learning_rate": 0.00045973645680819915, "epoch": 4.686744815148783, "step": 6500}, {"loss": 1.2121, "grad_norm": 0.3622857332229614, "learning_rate": 0.00045900439238653, "epoch": 4.758881875563571, "step": 6600}, {"loss": 1.2335, "grad_norm": 0.3443882465362549, "learning_rate": 0.0004582723279648609, "epoch": 4.831018935978359, "step": 6700}, {"loss": 1.2067, "grad_norm": 0.3408307433128357, "learning_rate": 0.0004575402635431918, "epoch": 4.9031559963931475, "step": 6800}, {"loss": 1.2229, "grad_norm": 1.0336593389511108, "learning_rate": 0.0004568081991215227, "epoch": 4.975293056807935, "step": 6900}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 4.0, "global_step": 5548, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 1.283941092094894e+18, "log_history": [{"loss": 1.3447, "grad_norm": 0.23201945424079895, "learning_rate": 0.00047657393850658857, "epoch": 3.028133453561767, "step": 4200}, {"loss": 1.2226, "grad_norm": 0.26054441928863525, "learning_rate": 0.0004758418740849195, "epoch": 3.1002705139765556, "step": 4300}, {"loss": 1.2542, "grad_norm": 0.38208144903182983, "learning_rate": 0.0004751098096632504, "epoch": 3.1724075743913436, "step": 4400}, {"loss": 1.32, "grad_norm": 0.32917675375938416, "learning_rate": 0.0004743777452415813, "epoch": 3.2445446348061315, "step": 4500}, {"loss": 1.2849, "grad_norm": 0.2643943727016449, "learning_rate": 0.0004736456808199122, "epoch": 3.31668169522092, "step": 4600}, {"loss": 1.2998, "grad_norm": 0.26708438992500305, "learning_rate": 0.00047291361639824304, "epoch": 3.388818755635708, "step": 4700}, {"loss": 1.2991, "grad_norm": 0.3026176989078522, "learning_rate": 0.000472181551976574, "epoch": 3.460955816050496, "step": 4800}, {"loss": 1.3209, "grad_norm": 0.2785812318325043, "learning_rate": 0.00047144948755490485, "epoch": 3.5330928764652842, "step": 4900}, {"loss": 1.3213, "grad_norm": 0.3397505283355713, "learning_rate": 0.00047071742313323575, "epoch": 3.605229936880072, "step": 5000}, {"loss": 1.3182, "grad_norm": 0.3531794846057892, "learning_rate": 0.0004699853587115666, "epoch": 3.67736699729486, "step": 5100}, {"loss": 1.3469, "grad_norm": 0.2614156901836395, "learning_rate": 0.00046925329428989756, "epoch": 3.7495040577096486, "step": 5200}, {"loss": 1.3348, "grad_norm": 0.25941088795661926, "learning_rate": 0.0004685212298682284, "epoch": 3.8216411181244365, "step": 5300}, {"loss": 1.3187, "grad_norm": 0.3109814524650574, "learning_rate": 0.0004677891654465593, "epoch": 3.8937781785392245, "step": 5400}, {"loss": 1.3378, "grad_norm": 0.3091738522052765, "learning_rate": 0.00046705710102489017, "epoch": 3.965915238954013, "step": 5500}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 3.0, "global_step": 4161, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 9.573211576144282e+17, "log_history": [{"loss": 1.5276, "grad_norm": 0.21804291009902954, "learning_rate": 0.0004868228404099561, "epoch": 2.018755635707845, "step": 2800}, {"loss": 1.4001, "grad_norm": 0.23674795031547546, "learning_rate": 0.000486090775988287, "epoch": 2.090892696122633, "step": 2900}, {"loss": 1.4131, "grad_norm": 0.27299487590789795, "learning_rate": 0.0004853587115666179, "epoch": 2.163029756537421, "step": 3000}, {"loss": 1.4123, "grad_norm": 0.23749138414859772, "learning_rate": 0.00048462664714494873, "epoch": 2.2351668169522094, "step": 3100}, {"loss": 1.4314, "grad_norm": 0.2305181324481964, "learning_rate": 0.0004838945827232797, "epoch": 2.3073038773669974, "step": 3200}, {"loss": 1.4345, "grad_norm": 0.28929394483566284, "learning_rate": 0.00048316251830161054, "epoch": 2.3794409377817853, "step": 3300}, {"loss": 1.4504, "grad_norm": 0.24168799817562103, "learning_rate": 0.00048243045387994144, "epoch": 2.4515779981965737, "step": 3400}, {"loss": 1.4092, "grad_norm": 0.22663119435310364, "learning_rate": 0.00048169838945827235, "epoch": 2.5237150586113617, "step": 3500}, {"loss": 1.4441, "grad_norm": 0.22306790947914124, "learning_rate": 0.00048096632503660325, "epoch": 2.5958521190261497, "step": 3600}, {"loss": 1.4398, "grad_norm": 0.24016821384429932, "learning_rate": 0.00048023426061493416, "epoch": 2.667989179440938, "step": 3700}, {"loss": 1.4247, "grad_norm": 0.2584359645843506, "learning_rate": 0.000479502196193265, "epoch": 2.740126239855726, "step": 3800}, {"loss": 1.4313, "grad_norm": 0.22991204261779785, "learning_rate": 0.0004787701317715959, "epoch": 2.812263300270514, "step": 3900}, {"loss": 1.4168, "grad_norm": 0.3092075288295746, "learning_rate": 0.0004780380673499268, "epoch": 2.884400360685302, "step": 4000}, {"loss": 1.4454, "grad_norm": 0.25069093704223633, "learning_rate": 0.0004773060029282577, "epoch": 2.9565374211000903, "step": 4100}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 2.0, "global_step": 2774, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 6.297918511632998e+17, "log_history": [{"loss": 1.6128, "grad_norm": 0.16140592098236084, "learning_rate": 0.0004970717423133236, "epoch": 1.0093778178539226, "step": 1400}, {"loss": 1.5563, "grad_norm": 0.15245527029037476, "learning_rate": 0.0004963396778916544, "epoch": 1.0815148782687105, "step": 1500}, {"loss": 1.5519, "grad_norm": 0.17794470489025116, "learning_rate": 0.0004956076134699854, "epoch": 1.1536519386834987, "step": 1600}, {"loss": 1.57, "grad_norm": 0.17832154035568237, "learning_rate": 0.0004948755490483162, "epoch": 1.2257889990982869, "step": 1700}, {"loss": 1.572, "grad_norm": 0.19502845406532288, "learning_rate": 0.0004941434846266471, "epoch": 1.2979260595130748, "step": 1800}, {"loss": 1.5587, "grad_norm": 0.1752757877111435, "learning_rate": 0.000493411420204978, "epoch": 1.370063119927863, "step": 1900}, {"loss": 1.5303, "grad_norm": 0.21198248863220215, "learning_rate": 0.0004926793557833089, "epoch": 1.442200180342651, "step": 2000}, {"loss": 1.5477, "grad_norm": 0.19046159088611603, "learning_rate": 0.0004919472913616399, "epoch": 1.5143372407574391, "step": 2100}, {"loss": 1.5618, "grad_norm": 0.17039573192596436, "learning_rate": 0.0004912152269399708, "epoch": 1.586474301172227, "step": 2200}, {"loss": 1.5347, "grad_norm": 0.2081579715013504, "learning_rate": 0.0004904831625183017, "epoch": 1.6586113615870153, "step": 2300}, {"loss": 1.5333, "grad_norm": 0.18298844993114471, "learning_rate": 0.0004897510980966326, "epoch": 1.7307484220018035, "step": 2400}, {"loss": 1.5208, "grad_norm": 0.19777992367744446, "learning_rate": 0.0004890190336749635, "epoch": 1.8028854824165914, "step": 2500}, {"loss": 1.5338, "grad_norm": 0.18976591527462006, "learning_rate": 0.0004882869692532943, "epoch": 1.8750225428313796, "step": 2600}, {"loss": 1.5353, "grad_norm": 0.19241571426391602, "learning_rate": 0.00048755490483162517, "epoch": 1.9471596032461678, "step": 2700}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 1.0, "global_step": 1387, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 3.030056675461325e+17, "log_history": [{"loss": 1.9362, "grad_norm": 0.1440935730934143, "learning_rate": 5e-05, "epoch": 0.0721370604147881, "step": 100}, {"loss": 1.7689, "grad_norm": 0.19405560195446014, "learning_rate": 0.0001, "epoch": 0.1442741208295762, "step": 200}, {"loss": 1.6634, "grad_norm": 0.1908496767282486, "learning_rate": 0.00015, "epoch": 0.2164111812443643, "step": 300}, {"loss": 1.6822, "grad_norm": 0.17500998079776764, "learning_rate": 0.0002, "epoch": 0.2885482416591524, "step": 400}, {"loss": 1.6399, "grad_norm": 0.1859809011220932, "learning_rate": 0.00025, "epoch": 0.3606853020739405, "step": 500}, {"loss": 1.6548, "grad_norm": 0.16745209693908691, "learning_rate": 0.0003, "epoch": 0.4328223624887286, "step": 600}, {"loss": 1.6067, "grad_norm": 0.17589329183101654, "learning_rate": 0.00035, "epoch": 0.5049594229035167, "step": 700}, {"loss": 1.6306, "grad_norm": 0.13673990964889526, "learning_rate": 0.0004, "epoch": 0.5770964833183048, "step": 800}, {"loss": 1.6005, "grad_norm": 0.2886543869972229, "learning_rate": 0.00045000000000000004, "epoch": 0.6492335437330928, "step": 900}, {"loss": 1.6252, "grad_norm": 0.15023812651634216, "learning_rate": 0.0005, "epoch": 0.721370604147881, "step": 1000}, {"loss": 1.6388, "grad_norm": 0.2001035511493683, "learning_rate": 0.0004992679355783309, "epoch": 0.7935076645626691, "step": 1100}, {"loss": 1.6283, "grad_norm": 0.17383413016796112, "learning_rate": 0.0004985358711566618, "epoch": 0.8656447249774571, "step": 1200}, {"loss": 1.5925, "grad_norm": 0.15884488821029663, "learning_rate": 0.0004978038067349927, "epoch": 0.9377817853922452, "step": 1300}], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": true, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 0, "global_step": 0, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 0, "log_history": [], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": false, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
{"epoch": 0, "global_step": 0, "max_steps": 69300, "logging_steps": 100, "eval_steps": 500, "save_steps": 500, "train_batch_size": 2, "num_train_epochs": 50, "num_input_tokens_seen": 0, "total_flos": 0, "log_history": [], "best_metric": null, "best_model_checkpoint": null, "is_local_process_zero": true, "is_world_process_zero": true, "is_hyper_param_search": false, "trial_name": null, "trial_params": null, "stateful_callbacks": {"TrainerControl": {"args": {"should_training_stop": false, "should_epoch_stop": false, "should_save": false, "should_evaluate": false, "should_log": false}, "attributes": {}}}} (Trained with Unsloth)
Upload README.md with huggingface_hub
initial commit
