LFM2.5-2.6B-UNCENSORED-ABLITERATED-PHILADELPHIA-CLASS / evals /matched_public_capability_lm_eval_0_4_12.json
Justbackup's picture
Duplicate from KridgeDookie/LFM2.5-2.6B-UNCENSORED-ABLITERATED-PHILADELPHIA-CLASS
970090c
Raw
History Blame Contribute Delete
4.98 kB
{
"protocol": {
"backend": "hf",
"batch_size": 8,
"chat_template": false,
"dtype": "bfloat16",
"fewshot": {
"gsm8k": 5,
"hellaswag": 0,
"mmlu": 0,
"truthfulqa_mc2": 0,
"winogrande": 0
},
"harness": {
"name": "lm-evaluation-harness",
"raw_result_git_hash": {
"candidate": null,
"note": "This per-result field is derived from the launch working directory, not the installed evaluator package. The upstream value is unavailable and is not inferred.",
"untouched_upstream": null,
"used_for_protocol_matching": false
},
"version": "0.4.12"
},
"limit": null,
"seeds": {
"fewshot_seed": 1234,
"numpy_seed": 1234,
"random_seed": 0,
"torch_seed": 1234
},
"system_instruction": false,
"task_versions": {
"gsm8k": 3.0,
"hellaswag": 1.0,
"mmlu": "2",
"truthfulqa_mc2": 3.0,
"winogrande": 1.0
},
"transformers_version": "5.12.1"
},
"schema_version": 1,
"scope": "matched_public_capability_evaluation",
"sources": {
"candidate_result_sha256": "a858842c8c9ce384621e92a54104d4d27baa00af09770eaad001ee74be60ae8a",
"untouched_upstream_result_sha256": "42542d992069b03084e2c5afb03b39fd54f4551878391976a79a3b0f774acf35"
},
"tasks": {
"gsm8k": {
"interpretation": "Generated-answer exact match with strict extraction is primary; flexible extraction is secondary. Higher is better.",
"metrics": {
"exact_match,flexible-extract": {
"candidate": {
"stderr": 0.012757375376754941,
"value": 0.6884003032600455
},
"delta_candidate_minus_upstream": -0.010614101592115177,
"role": "secondary",
"untouched_upstream": {
"stderr": 0.012634504465211183,
"value": 0.6990144048521607
}
},
"exact_match,strict-match": {
"candidate": {
"stderr": 0.01268813407672688,
"value": 0.6944655041698257
},
"delta_candidate_minus_upstream": -0.0037907505686125553,
"role": "primary",
"untouched_upstream": {
"stderr": 0.012643544762873356,
"value": 0.6982562547384382
}
}
},
"sample_count": 1319
},
"hellaswag": {
"interpretation": "Length-normalized multiple-choice accuracy is primary; raw accuracy is secondary. Higher is better.",
"metrics": {
"acc,none": {
"candidate": {
"stderr": 0.004954146286513348,
"value": 0.4403505277833101
},
"delta_candidate_minus_upstream": 0.0008962358095996326,
"role": "secondary",
"untouched_upstream": {
"stderr": 0.004953063404791443,
"value": 0.43945429197371044
}
},
"acc_norm,none": {
"candidate": {
"stderr": 0.004940631135803536,
"value": 0.5700059749053973
},
"delta_candidate_minus_upstream": 0.002887870942043347,
"role": "primary",
"untouched_upstream": {
"stderr": 0.004944620712318271,
"value": 0.5671181039633539
}
}
},
"sample_count": 10042
},
"mmlu": {
"interpretation": "Accuracy over all MMLU subjects; higher is better.",
"metrics": {
"acc,none": {
"candidate": {
"stderr": 0.0035976362847934605,
"value": 0.2404928072924085
},
"delta_candidate_minus_upstream": 0.0024213075060532663,
"role": "primary",
"untouched_upstream": {
"stderr": 0.0035855541877085647,
"value": 0.23807149978635522
}
}
},
"sample_count": 14042
},
"truthfulqa_mc2": {
"interpretation": "Mean normalized probability mass assigned to true answer options; higher is better.",
"metrics": {
"acc,none": {
"candidate": {
"stderr": 0.015423514306336993,
"value": 0.5378683032141651
},
"delta_candidate_minus_upstream": -0.024131181526524492,
"role": "primary",
"untouched_upstream": {
"stderr": 0.015309998030227038,
"value": 0.5619994847406896
}
}
},
"sample_count": 817
},
"winogrande": {
"interpretation": "Multiple-choice accuracy; higher is better.",
"metrics": {
"acc,none": {
"candidate": {
"stderr": 0.01375574351374902,
"value": 0.6022099447513812
},
"delta_candidate_minus_upstream": -0.0023677979479084232,
"role": "primary",
"untouched_upstream": {
"stderr": 0.013741678387545354,
"value": 0.6045777426992897
}
}
},
"sample_count": 1267
}
}
}