{ "best_global_step": 4000, "best_metric": 37.39491327019262, "best_model_checkpoint": "./whisper-medium-ru-lora/checkpoint-4000", "epoch": 0.30328303889604974, "eval_steps": 1000, "global_step": 4000, "is_hyper_param_search": false, "is_local_process_zero": true, "is_world_process_zero": true, "log_history": [ { "epoch": 0.0075820759724012435, "grad_norm": 2.145202398300171, "learning_rate": 9.900000000000002e-06, "loss": 0.651605224609375, "step": 100 }, { "epoch": 0.015164151944802487, "grad_norm": 1.6405466794967651, "learning_rate": 1.9900000000000003e-05, "loss": 0.38233642578125, "step": 200 }, { "epoch": 0.02274622791720373, "grad_norm": 4.584617614746094, "learning_rate": 2.9900000000000002e-05, "loss": 0.15616183280944823, "step": 300 }, { "epoch": 0.030328303889604974, "grad_norm": 1.5751986503601074, "learning_rate": 3.99e-05, "loss": 0.1865867042541504, "step": 400 }, { "epoch": 0.03791037986200622, "grad_norm": 2.414318561553955, "learning_rate": 4.99e-05, "loss": 0.17033681869506836, "step": 500 }, { "epoch": 0.04549245583440746, "grad_norm": 2.5033624172210693, "learning_rate": 4.858571428571429e-05, "loss": 0.15961559295654296, "step": 600 }, { "epoch": 0.053074531806808704, "grad_norm": 2.129359245300293, "learning_rate": 4.715714285714286e-05, "loss": 0.16227605819702148, "step": 700 }, { "epoch": 0.06065660777920995, "grad_norm": 0.5429865717887878, "learning_rate": 4.572857142857143e-05, "loss": 0.17982370376586915, "step": 800 }, { "epoch": 0.06823868375161118, "grad_norm": 0.09488433599472046, "learning_rate": 4.43e-05, "loss": 0.15608755111694336, "step": 900 }, { "epoch": 0.07582075972401243, "grad_norm": 1.4993505477905273, "learning_rate": 4.287142857142857e-05, "loss": 0.15827343940734864, "step": 1000 }, { "epoch": 0.07582075972401243, "eval_loss": 0.1047014594078064, "eval_runtime": 1260.1346, "eval_samples_per_second": 0.794, "eval_steps_per_second": 0.397, "eval_wer": 38.26753219112483, "step": 1000 }, { "epoch": 0.08340283569641367, "grad_norm": 0.5261468291282654, "learning_rate": 4.1442857142857146e-05, "loss": 0.1773993492126465, "step": 1100 }, { "epoch": 0.09098491166881492, "grad_norm": 3.5084214210510254, "learning_rate": 4.001428571428571e-05, "loss": 0.13967780113220216, "step": 1200 }, { "epoch": 0.09856698764121616, "grad_norm": 2.6616625785827637, "learning_rate": 3.8585714285714287e-05, "loss": 0.1545235538482666, "step": 1300 }, { "epoch": 0.10614906361361741, "grad_norm": 2.668151378631592, "learning_rate": 3.715714285714286e-05, "loss": 0.15973119735717772, "step": 1400 }, { "epoch": 0.11373113958601865, "grad_norm": 0.329638808965683, "learning_rate": 3.572857142857143e-05, "loss": 0.12528255462646484, "step": 1500 }, { "epoch": 0.1213132155584199, "grad_norm": 0.5731531381607056, "learning_rate": 3.430000000000001e-05, "loss": 0.12541525840759277, "step": 1600 }, { "epoch": 0.12889529153082113, "grad_norm": 1.0524909496307373, "learning_rate": 3.2871428571428574e-05, "loss": 0.14487401962280275, "step": 1700 }, { "epoch": 0.13647736750322237, "grad_norm": 0.17586998641490936, "learning_rate": 3.144285714285715e-05, "loss": 0.11433093070983887, "step": 1800 }, { "epoch": 0.14405944347562363, "grad_norm": 1.013460636138916, "learning_rate": 3.0014285714285717e-05, "loss": 0.15009353637695313, "step": 1900 }, { "epoch": 0.15164151944802487, "grad_norm": 1.2082782983779907, "learning_rate": 2.8585714285714287e-05, "loss": 0.15209393501281737, "step": 2000 }, { "epoch": 0.15164151944802487, "eval_loss": 0.09539687633514404, "eval_runtime": 1270.093, "eval_samples_per_second": 0.787, "eval_steps_per_second": 0.394, "eval_wer": 37.83122273065872, "step": 2000 }, { "epoch": 0.1592235954204261, "grad_norm": 0.7301309108734131, "learning_rate": 2.7157142857142858e-05, "loss": 0.12468180656433106, "step": 2100 }, { "epoch": 0.16680567139282734, "grad_norm": 1.471038579940796, "learning_rate": 2.572857142857143e-05, "loss": 0.13947233200073242, "step": 2200 }, { "epoch": 0.1743877473652286, "grad_norm": 3.3145599365234375, "learning_rate": 2.43e-05, "loss": 0.15013423919677735, "step": 2300 }, { "epoch": 0.18196982333762984, "grad_norm": 1.1346559524536133, "learning_rate": 2.287142857142857e-05, "loss": 0.13823192596435546, "step": 2400 }, { "epoch": 0.18955189931003108, "grad_norm": 0.8875208497047424, "learning_rate": 2.1442857142857145e-05, "loss": 0.13282781600952148, "step": 2500 }, { "epoch": 0.19713397528243232, "grad_norm": 1.3990944623947144, "learning_rate": 2.0014285714285715e-05, "loss": 0.15458744049072265, "step": 2600 }, { "epoch": 0.20471605125483358, "grad_norm": 4.13588809967041, "learning_rate": 1.858571428571429e-05, "loss": 0.1336074447631836, "step": 2700 }, { "epoch": 0.21229812722723482, "grad_norm": 3.2681021690368652, "learning_rate": 1.715714285714286e-05, "loss": 0.14435559272766113, "step": 2800 }, { "epoch": 0.21988020319963605, "grad_norm": 2.823951244354248, "learning_rate": 1.572857142857143e-05, "loss": 0.13421536445617677, "step": 2900 }, { "epoch": 0.2274622791720373, "grad_norm": 1.4937087297439575, "learning_rate": 1.43e-05, "loss": 0.12853222846984863, "step": 3000 }, { "epoch": 0.2274622791720373, "eval_loss": 0.09113000333309174, "eval_runtime": 1257.6368, "eval_samples_per_second": 0.795, "eval_steps_per_second": 0.398, "eval_wer": 37.639672235819944, "step": 3000 }, { "epoch": 0.23504435514443855, "grad_norm": 0.5249112844467163, "learning_rate": 1.2871428571428574e-05, "loss": 0.1645351791381836, "step": 3100 }, { "epoch": 0.2426264311168398, "grad_norm": 4.014407634735107, "learning_rate": 1.1442857142857144e-05, "loss": 0.14470401763916016, "step": 3200 }, { "epoch": 0.25020850708924103, "grad_norm": 0.6329492330551147, "learning_rate": 1.0014285714285716e-05, "loss": 0.12239849090576171, "step": 3300 }, { "epoch": 0.25779058306164226, "grad_norm": 2.205402135848999, "learning_rate": 8.585714285714286e-06, "loss": 0.15601092338562011, "step": 3400 }, { "epoch": 0.2653726590340435, "grad_norm": 3.0992422103881836, "learning_rate": 7.1571428571428584e-06, "loss": 0.1359235954284668, "step": 3500 }, { "epoch": 0.27295473500644474, "grad_norm": 1.166153073310852, "learning_rate": 5.728571428571429e-06, "loss": 0.1488318347930908, "step": 3600 }, { "epoch": 0.28053681097884603, "grad_norm": 0.10292654484510422, "learning_rate": 4.2999999999999995e-06, "loss": 0.1741475486755371, "step": 3700 }, { "epoch": 0.28811888695124727, "grad_norm": 0.1904854029417038, "learning_rate": 2.8714285714285713e-06, "loss": 0.12981878280639647, "step": 3800 }, { "epoch": 0.2957009629236485, "grad_norm": 1.543544888496399, "learning_rate": 1.4428571428571429e-06, "loss": 0.16215293884277343, "step": 3900 }, { "epoch": 0.30328303889604974, "grad_norm": 1.1929818391799927, "learning_rate": 1.4285714285714288e-08, "loss": 0.14198240280151367, "step": 4000 }, { "epoch": 0.30328303889604974, "eval_loss": 0.08855394273996353, "eval_runtime": 1257.4555, "eval_samples_per_second": 0.795, "eval_steps_per_second": 0.398, "eval_wer": 37.39491327019262, "step": 4000 } ], "logging_steps": 100, "max_steps": 4000, "num_input_tokens_seen": 0, "num_train_epochs": 1, "save_steps": 1000, "stateful_callbacks": { "TrainerControl": { "args": { "should_epoch_stop": false, "should_evaluate": false, "should_log": false, "should_save": true, "should_training_stop": true }, "attributes": {} } }, "total_flos": 8.38227197952e+18, "train_batch_size": 2, "trial_name": null, "trial_params": null }