Text Generation
Transformers
Safetensors
abstract-cot
latent-reasoning
math-reasoning
qwen3
leapeto's picture
Add files using upload-large-folder tool
a555798 verified
{
"losses": [
0.35318275502650065,
0.30725667767983394,
0.34089534296654167,
0.3377619634848088,
0.3068910479894839,
0.3112259633140638,
0.35649422819260507,
0.3305382497259416,
0.3459838313516229,
0.3343647971516475,
0.33608090435154736,
0.3339404923695838,
0.32868876951106357,
0.364680266357027,
0.31102049429900946,
0.3481577365193516,
0.3247197750257328,
0.33266925923526286,
0.35477414953056724,
0.33790302132838407,
0.35382214419078084,
0.36831805281108243,
0.3632078022346832,
0.34023994151502845,
0.3407576463650912,
0.40222772396809886,
0.3644849105272442,
0.31388317197561266,
0.36352235367521646,
0.35906730571296064,
0.34190756385796706
],
"lrs": [
6.666666666666667e-05,
9.993008576227247e-05,
9.937194443381972e-05,
9.826190093588563e-05,
9.661236384224129e-05,
9.444177243274618e-05,
9.177439057064683e-05,
8.864003547001915e-05,
8.507374438531607e-05,
8.111538294891684e-05,
7.680919953486048e-05,
7.220333063028872e-05,
6.734926274378312e-05,
6.230125686563068e-05,
5.7115741913664264e-05,
5.185068394501791e-05,
4.6564938185035956e-05,
4.131759111665349e-05,
3.616729998467365e-05,
3.1171637098265064e-05,
2.638644626136587e-05,
2.1865218525109495e-05,
1.7658494240397126e-05,
1.3813298094746491e-05,
1.037261344883343e-05,
7.374901848832683e-06,
4.853673085668947e-06,
2.8371106072518195e-06,
1.3477564710088098e-06,
4.02259358460233e-07,
1.1188468644907079e-08
],
"wallclock_s": 10542,
"n_examples": 5000,
"epochs": 1,
"mode": "bottleneck",
"lora_rank": 32,
"total_opt_steps": 156,
"num_processes": 2
}