CoolFace
Modelpublic

Kark07/dapo_phi3.5_code

sourceHugging Faceupdated 8mo agoView on Hugging Face
0likes8downloads
trainer_state.json192 linesDownload Raw Back to checkpoint-50
1{2  "best_global_step": null,3  "best_metric": null,4  "best_model_checkpoint": null,5  "epoch": 2.0,6  "eval_steps": 100,7  "global_step": 50,8  "is_hyper_param_search": false,9  "is_local_process_zero": true,10  "is_world_process_zero": true,11  "log_history": [12    {13      "clip_ratio/high_max": 0.0,14      "clip_ratio/high_mean": 0.0,15      "clip_ratio/low_mean": 0.0,16      "clip_ratio/low_min": 0.0,17      "clip_ratio/region_mean": 0.0,18      "completions/clipped_ratio": 0.8671875,19      "completions/max_length": 256.0,20      "completions/max_terminated_length": 244.0,21      "completions/mean_length": 244.26015625,22      "completions/mean_terminated_length": 170.32069549560546,23      "completions/min_length": 62.2,24      "completions/min_terminated_length": 62.2,25      "entropy": 0.5070019140839577,26      "epoch": 0.4,27      "frac_reward_zero_std": 0.0,28      "grad_norm": 0.011515798047184944,29      "learning_rate": 9e-07,30      "loss": -0.0037,31      "num_tokens": 337617.0,32      "reward": 0.3755455553531647,33      "reward_std": 0.06347588822245598,34      "rewards/_combined_reward_function/mean": 0.3755455493927002,35      "rewards/_combined_reward_function/std": 0.09810434952378273,36      "step": 1037    },38    {39      "clip_ratio/high_max": 0.0,40      "clip_ratio/high_mean": 0.0,41      "clip_ratio/low_mean": 0.0,42      "clip_ratio/low_min": 0.0,43      "clip_ratio/region_mean": 0.0,44      "completions/clipped_ratio": 0.84140625,45      "completions/max_length": 256.0,46      "completions/max_terminated_length": 250.3,47      "completions/mean_length": 241.9859375,48      "completions/mean_terminated_length": 166.67403869628907,49      "completions/min_length": 54.7,50      "completions/min_terminated_length": 54.7,51      "entropy": 0.5022411059588194,52      "epoch": 0.8,53      "frac_reward_zero_std": 0.0,54      "grad_norm": 0.01148963626474142,55      "learning_rate": 8.615384615384616e-07,56      "loss": -0.002,57      "num_tokens": 672379.0,58      "reward": 0.3655878365039825,59      "reward_std": 0.05772071108222008,60      "rewards/_combined_reward_function/mean": 0.3655878394842148,61      "rewards/_combined_reward_function/std": 0.09406096115708351,62      "step": 2063    },64    {65      "epoch": 1.0,66      "eval_clip_ratio/high_max": 0.0,67      "eval_clip_ratio/high_mean": 0.0,68      "eval_clip_ratio/low_mean": 0.0,69      "eval_clip_ratio/low_min": 0.0,70      "eval_clip_ratio/region_mean": 0.0,71      "eval_completions/clipped_ratio": 0.9075,72      "eval_completions/max_length": 255.78,73      "eval_completions/max_terminated_length": 53.59,74      "eval_completions/mean_length": 249.0225,75      "eval_completions/mean_terminated_length": 50.3025,76      "eval_completions/min_length": 233.77,77      "eval_completions/min_terminated_length": 46.89,78      "eval_entropy": 0.49880172580480575,79      "eval_frac_reward_zero_std": 0.0,80      "eval_loss": 0.0012736087664961815,81      "eval_num_tokens": 840273.0,82      "eval_reward": 0.37732040256261823,83      "eval_reward_std": 0.0592539354134351,84      "eval_rewards/_combined_reward_function/mean": 0.37732040256261823,85      "eval_rewards/_combined_reward_function/std": 0.05925393763929605,86      "eval_runtime": 1175.6812,87      "eval_samples_per_second": 0.085,88      "eval_steps_per_second": 0.021,89      "step": 2590    },91    {92      "clip_ratio/high_max": 0.0,93      "clip_ratio/high_mean": 0.0,94      "clip_ratio/low_mean": 0.0,95      "clip_ratio/low_min": 0.0,96      "clip_ratio/region_mean": 0.0,97      "completions/clipped_ratio": 0.8453125,98      "completions/max_length": 256.0,99      "completions/max_terminated_length": 251.9,100      "completions/mean_length": 242.99296875,101      "completions/mean_terminated_length": 172.87078704833985,102      "completions/min_length": 65.0,103      "completions/min_terminated_length": 65.0,104      "entropy": 0.49796839645132424,105      "epoch": 1.2,106      "frac_reward_zero_std": 0.0,107      "grad_norm": 0.010568513534963131,108      "learning_rate": 7.076923076923077e-07,109      "loss": -0.0081,110      "num_tokens": 1009346.0,111      "reward": 0.3748364597558975,112      "reward_std": 0.05954046621918678,113      "rewards/_combined_reward_function/mean": 0.3748364597558975,114      "rewards/_combined_reward_function/std": 0.09809738993644715,115      "step": 30116    },117    {118      "clip_ratio/high_max": 0.0,119      "clip_ratio/high_mean": 0.0,120      "clip_ratio/low_mean": 0.0,121      "clip_ratio/low_min": 0.0,122      "clip_ratio/region_mean": 0.0,123      "completions/clipped_ratio": 0.84609375,124      "completions/max_length": 256.0,125      "completions/max_terminated_length": 250.0,126      "completions/mean_length": 242.29140625,127      "completions/mean_terminated_length": 168.146484375,128      "completions/min_length": 65.5,129      "completions/min_terminated_length": 65.5,130      "entropy": 0.4996831588447094,131      "epoch": 1.6,132      "frac_reward_zero_std": 0.0,133      "grad_norm": 0.011787918396294117,134      "learning_rate": 5.538461538461539e-07,135      "loss": 0.0045,136      "num_tokens": 1344259.0,137      "reward": 0.37461447417736055,138      "reward_std": 0.061694014072418216,139      "rewards/_combined_reward_function/mean": 0.37461447417736055,140      "rewards/_combined_reward_function/std": 0.09879581183195114,141      "step": 40142    },143    {144      "clip_ratio/high_max": 0.0,145      "clip_ratio/high_mean": 0.0,146      "clip_ratio/low_mean": 0.0,147      "clip_ratio/low_min": 0.0,148      "clip_ratio/region_mean": 0.0,149      "completions/clipped_ratio": 0.871875,150      "completions/max_length": 256.0,151      "completions/max_terminated_length": 242.6,152      "completions/mean_length": 244.01484375,153      "completions/mean_terminated_length": 161.6182373046875,154      "completions/min_length": 44.5,155      "completions/min_terminated_length": 44.5,156      "entropy": 0.5059020014014095,157      "epoch": 2.0,158      "frac_reward_zero_std": 0.0,159      "grad_norm": 0.011442111805081367,160      "learning_rate": 4e-07,161      "loss": 0.0007,162      "num_tokens": 1681842.0,163      "reward": 0.37922771871089933,164      "reward_std": 0.06254741400480271,165      "rewards/_combined_reward_function/mean": 0.37922771871089933,166      "rewards/_combined_reward_function/std": 0.1040997788310051,167      "step": 50168    }169  ],170  "logging_steps": 10,171  "max_steps": 75,172  "num_input_tokens_seen": 1681842,173  "num_train_epochs": 3,174  "save_steps": 50,175  "stateful_callbacks": {176    "TrainerControl": {177      "args": {178        "should_epoch_stop": false,179        "should_evaluate": false,180        "should_log": false,181        "should_save": true,182        "should_training_stop": false183      },184      "attributes": {}185    }186  },187  "total_flos": 0.0,188  "train_batch_size": 4,189  "trial_name": null,190  "trial_params": null191}192