kacperwikiel's picture
Release experimental E01 SFT with verified before-after benchmarks
908ce20 verified
Raw
History Blame Contribute Delete
1.96 kB
{
"complete": true,
"categories": {
"code_completion": {
"before_correct": 5,
"after_correct": 6,
"gained": 3,
"lost": 2,
"paired_exact_p_uncorrected": 1.0,
"before_elo": 729,
"after_elo": 724
},
"commonsense": {
"before_correct": 15,
"after_correct": 18,
"gained": 4,
"lost": 1,
"paired_exact_p_uncorrected": 0.375,
"before_elo": 772,
"after_elo": 815
},
"context_tracking": {
"before_correct": 11,
"after_correct": 15,
"gained": 7,
"lost": 3,
"paired_exact_p_uncorrected": 0.34375,
"before_elo": 740,
"after_elo": 838
},
"language_completion": {
"before_correct": 30,
"after_correct": 31,
"gained": 4,
"lost": 3,
"paired_exact_p_uncorrected": 1.0,
"before_elo": 993,
"after_elo": 1015
},
"logical_reasoning": {
"before_correct": 14,
"after_correct": 16,
"gained": 5,
"lost": 3,
"paired_exact_p_uncorrected": 0.7265625,
"before_elo": 939,
"after_elo": 960
},
"quantitative": {
"before_correct": 16,
"after_correct": 9,
"gained": 1,
"lost": 8,
"paired_exact_p_uncorrected": 0.0390625,
"before_elo": 925,
"after_elo": 788
},
"world_knowledge": {
"before_correct": 15,
"after_correct": 18,
"gained": 6,
"lost": 3,
"paired_exact_p_uncorrected": 0.5078125,
"before_elo": 791,
"after_elo": 840
}
},
"overall_elo_before": 842,
"overall_elo_after": 853,
"limitations": [
"Base continuation benchmark, not executable coding or Instruct Bench.",
"E01 used CPU FP32; SFT used CUDA FP32.",
"Small category sample sizes; uncorrected paired tests are descriptive."
],
"conclusion": "No convincing coding improvement: one extra correct code item, lower code Elo, and quantitative regression."
}