272 lines
7.0 KiB
JSON
272 lines
7.0 KiB
JSON
{
|
|
"best_global_step": 3000,
|
|
"best_metric": 40.58742151750558,
|
|
"best_model_checkpoint": "./whisper-small-ru-lora/checkpoint-3000",
|
|
"epoch": 0.2274622791720373,
|
|
"eval_steps": 1000,
|
|
"global_step": 3000,
|
|
"is_hyper_param_search": false,
|
|
"is_local_process_zero": true,
|
|
"is_world_process_zero": true,
|
|
"log_history": [
|
|
{
|
|
"epoch": 0.0075820759724012435,
|
|
"grad_norm": 2.718125820159912,
|
|
"learning_rate": 9.900000000000001e-05,
|
|
"loss": 0.5047353363037109,
|
|
"step": 100
|
|
},
|
|
{
|
|
"epoch": 0.015164151944802487,
|
|
"grad_norm": 2.834160089492798,
|
|
"learning_rate": 9.746153846153847e-05,
|
|
"loss": 0.26470003128051756,
|
|
"step": 200
|
|
},
|
|
{
|
|
"epoch": 0.02274622791720373,
|
|
"grad_norm": 5.67031192779541,
|
|
"learning_rate": 9.48974358974359e-05,
|
|
"loss": 0.24750186920166015,
|
|
"step": 300
|
|
},
|
|
{
|
|
"epoch": 0.030328303889604974,
|
|
"grad_norm": 2.8004355430603027,
|
|
"learning_rate": 9.233333333333333e-05,
|
|
"loss": 0.26615495681762696,
|
|
"step": 400
|
|
},
|
|
{
|
|
"epoch": 0.03791037986200622,
|
|
"grad_norm": 2.2990570068359375,
|
|
"learning_rate": 8.976923076923078e-05,
|
|
"loss": 0.2603847312927246,
|
|
"step": 500
|
|
},
|
|
{
|
|
"epoch": 0.04549245583440746,
|
|
"grad_norm": 2.2013652324676514,
|
|
"learning_rate": 8.720512820512821e-05,
|
|
"loss": 0.2441580390930176,
|
|
"step": 600
|
|
},
|
|
{
|
|
"epoch": 0.053074531806808704,
|
|
"grad_norm": 3.80888032913208,
|
|
"learning_rate": 8.464102564102564e-05,
|
|
"loss": 0.25420652389526366,
|
|
"step": 700
|
|
},
|
|
{
|
|
"epoch": 0.06065660777920995,
|
|
"grad_norm": 1.1088988780975342,
|
|
"learning_rate": 8.207692307692309e-05,
|
|
"loss": 0.2694035530090332,
|
|
"step": 800
|
|
},
|
|
{
|
|
"epoch": 0.06823868375161118,
|
|
"grad_norm": 0.9152566194534302,
|
|
"learning_rate": 7.951282051282052e-05,
|
|
"loss": 0.25536279678344725,
|
|
"step": 900
|
|
},
|
|
{
|
|
"epoch": 0.07582075972401243,
|
|
"grad_norm": 1.8358956575393677,
|
|
"learning_rate": 7.694871794871796e-05,
|
|
"loss": 0.23399829864501953,
|
|
"step": 1000
|
|
},
|
|
{
|
|
"epoch": 0.07582075972401243,
|
|
"eval_loss": 0.16988210380077362,
|
|
"eval_runtime": 400.9219,
|
|
"eval_samples_per_second": 2.494,
|
|
"eval_steps_per_second": 1.247,
|
|
"eval_wer": 41.51324890922635,
|
|
"step": 1000
|
|
},
|
|
{
|
|
"epoch": 0.08340283569641367,
|
|
"grad_norm": 2.426520586013794,
|
|
"learning_rate": 7.438461538461539e-05,
|
|
"loss": 0.2634130096435547,
|
|
"step": 1100
|
|
},
|
|
{
|
|
"epoch": 0.09098491166881492,
|
|
"grad_norm": 4.430180072784424,
|
|
"learning_rate": 7.182051282051282e-05,
|
|
"loss": 0.2219495964050293,
|
|
"step": 1200
|
|
},
|
|
{
|
|
"epoch": 0.09856698764121616,
|
|
"grad_norm": 2.884838104248047,
|
|
"learning_rate": 6.925641025641027e-05,
|
|
"loss": 0.22977264404296874,
|
|
"step": 1300
|
|
},
|
|
{
|
|
"epoch": 0.10614906361361741,
|
|
"grad_norm": 1.6225945949554443,
|
|
"learning_rate": 6.66923076923077e-05,
|
|
"loss": 0.24415328979492187,
|
|
"step": 1400
|
|
},
|
|
{
|
|
"epoch": 0.11373113958601865,
|
|
"grad_norm": 1.1895921230316162,
|
|
"learning_rate": 6.412820512820513e-05,
|
|
"loss": 0.1988828468322754,
|
|
"step": 1500
|
|
},
|
|
{
|
|
"epoch": 0.1213132155584199,
|
|
"grad_norm": 0.6518490314483643,
|
|
"learning_rate": 6.156410256410256e-05,
|
|
"loss": 0.208972225189209,
|
|
"step": 1600
|
|
},
|
|
{
|
|
"epoch": 0.12889529153082113,
|
|
"grad_norm": 1.2791333198547363,
|
|
"learning_rate": 5.9e-05,
|
|
"loss": 0.22393619537353515,
|
|
"step": 1700
|
|
},
|
|
{
|
|
"epoch": 0.13647736750322237,
|
|
"grad_norm": 0.653361976146698,
|
|
"learning_rate": 5.643589743589743e-05,
|
|
"loss": 0.1963739013671875,
|
|
"step": 1800
|
|
},
|
|
{
|
|
"epoch": 0.14405944347562363,
|
|
"grad_norm": 2.1832966804504395,
|
|
"learning_rate": 5.3871794871794876e-05,
|
|
"loss": 0.2256583595275879,
|
|
"step": 1900
|
|
},
|
|
{
|
|
"epoch": 0.15164151944802487,
|
|
"grad_norm": 1.687814712524414,
|
|
"learning_rate": 5.130769230769231e-05,
|
|
"loss": 0.22346633911132813,
|
|
"step": 2000
|
|
},
|
|
{
|
|
"epoch": 0.15164151944802487,
|
|
"eval_loss": 0.15810276567935944,
|
|
"eval_runtime": 399.4682,
|
|
"eval_samples_per_second": 2.503,
|
|
"eval_steps_per_second": 1.252,
|
|
"eval_wer": 41.26848994359902,
|
|
"step": 2000
|
|
},
|
|
{
|
|
"epoch": 0.1592235954204261,
|
|
"grad_norm": 2.3399932384490967,
|
|
"learning_rate": 4.874358974358975e-05,
|
|
"loss": 0.21493404388427734,
|
|
"step": 2100
|
|
},
|
|
{
|
|
"epoch": 0.16680567139282734,
|
|
"grad_norm": 3.332096815109253,
|
|
"learning_rate": 4.617948717948718e-05,
|
|
"loss": 0.21836400985717774,
|
|
"step": 2200
|
|
},
|
|
{
|
|
"epoch": 0.1743877473652286,
|
|
"grad_norm": 3.106832981109619,
|
|
"learning_rate": 4.361538461538462e-05,
|
|
"loss": 0.2281314277648926,
|
|
"step": 2300
|
|
},
|
|
{
|
|
"epoch": 0.18196982333762984,
|
|
"grad_norm": 3.9020333290100098,
|
|
"learning_rate": 4.105128205128205e-05,
|
|
"loss": 0.20598222732543944,
|
|
"step": 2400
|
|
},
|
|
{
|
|
"epoch": 0.18955189931003108,
|
|
"grad_norm": 2.090578079223633,
|
|
"learning_rate": 3.8487179487179486e-05,
|
|
"loss": 0.2045596122741699,
|
|
"step": 2500
|
|
},
|
|
{
|
|
"epoch": 0.19713397528243232,
|
|
"grad_norm": 3.6380014419555664,
|
|
"learning_rate": 3.5923076923076925e-05,
|
|
"loss": 0.23484689712524415,
|
|
"step": 2600
|
|
},
|
|
{
|
|
"epoch": 0.20471605125483358,
|
|
"grad_norm": 3.832679510116577,
|
|
"learning_rate": 3.335897435897436e-05,
|
|
"loss": 0.20866182327270508,
|
|
"step": 2700
|
|
},
|
|
{
|
|
"epoch": 0.21229812722723482,
|
|
"grad_norm": 4.924009323120117,
|
|
"learning_rate": 3.07948717948718e-05,
|
|
"loss": 0.2183872413635254,
|
|
"step": 2800
|
|
},
|
|
{
|
|
"epoch": 0.21988020319963605,
|
|
"grad_norm": 3.0527515411376953,
|
|
"learning_rate": 2.8230769230769233e-05,
|
|
"loss": 0.22305212020874024,
|
|
"step": 2900
|
|
},
|
|
{
|
|
"epoch": 0.2274622791720373,
|
|
"grad_norm": 1.3805994987487793,
|
|
"learning_rate": 2.5666666666666666e-05,
|
|
"loss": 0.22107456207275392,
|
|
"step": 3000
|
|
},
|
|
{
|
|
"epoch": 0.2274622791720373,
|
|
"eval_loss": 0.15330542623996735,
|
|
"eval_runtime": 398.709,
|
|
"eval_samples_per_second": 2.508,
|
|
"eval_steps_per_second": 1.254,
|
|
"eval_wer": 40.58742151750558,
|
|
"step": 3000
|
|
}
|
|
],
|
|
"logging_steps": 100,
|
|
"max_steps": 4000,
|
|
"num_input_tokens_seen": 0,
|
|
"num_train_epochs": 1,
|
|
"save_steps": 1000,
|
|
"stateful_callbacks": {
|
|
"TrainerControl": {
|
|
"args": {
|
|
"should_epoch_stop": false,
|
|
"should_evaluate": false,
|
|
"should_log": false,
|
|
"should_save": true,
|
|
"should_training_stop": false
|
|
},
|
|
"attributes": {}
|
|
}
|
|
},
|
|
"total_flos": 1.79266535424e+18,
|
|
"train_batch_size": 2,
|
|
"trial_name": null,
|
|
"trial_params": null
|
|
}
|