{ "protocol": { "backend": "hf", "batch_size": 8, "chat_template": false, "dtype": "bfloat16", "fewshot": { "gsm8k": 5, "hellaswag": 0, "mmlu": 0, "truthfulqa_mc2": 0, "winogrande": 0 }, "harness": { "name": "lm-evaluation-harness", "raw_result_git_hash": { "candidate": null, "note": "This per-result field is derived from the launch working directory, not the installed evaluator package. The upstream value is unavailable and is not inferred.", "untouched_upstream": null, "used_for_protocol_matching": false }, "version": "0.4.12" }, "limit": null, "seeds": { "fewshot_seed": 1234, "numpy_seed": 1234, "random_seed": 0, "torch_seed": 1234 }, "system_instruction": false, "task_versions": { "gsm8k": 3.0, "hellaswag": 1.0, "mmlu": "2", "truthfulqa_mc2": 3.0, "winogrande": 1.0 }, "transformers_version": "5.12.1" }, "schema_version": 1, "scope": "matched_public_capability_evaluation", "sources": { "candidate_result_sha256": "a858842c8c9ce384621e92a54104d4d27baa00af09770eaad001ee74be60ae8a", "untouched_upstream_result_sha256": "42542d992069b03084e2c5afb03b39fd54f4551878391976a79a3b0f774acf35" }, "tasks": { "gsm8k": { "interpretation": "Generated-answer exact match with strict extraction is primary; flexible extraction is secondary. Higher is better.", "metrics": { "exact_match,flexible-extract": { "candidate": { "stderr": 0.012757375376754941, "value": 0.6884003032600455 }, "delta_candidate_minus_upstream": -0.010614101592115177, "role": "secondary", "untouched_upstream": { "stderr": 0.012634504465211183, "value": 0.6990144048521607 } }, "exact_match,strict-match": { "candidate": { "stderr": 0.01268813407672688, "value": 0.6944655041698257 }, "delta_candidate_minus_upstream": -0.0037907505686125553, "role": "primary", "untouched_upstream": { "stderr": 0.012643544762873356, "value": 0.6982562547384382 } } }, "sample_count": 1319 }, "hellaswag": { "interpretation": "Length-normalized multiple-choice accuracy is primary; raw accuracy is secondary. Higher is better.", "metrics": { "acc,none": { "candidate": { "stderr": 0.004954146286513348, "value": 0.4403505277833101 }, "delta_candidate_minus_upstream": 0.0008962358095996326, "role": "secondary", "untouched_upstream": { "stderr": 0.004953063404791443, "value": 0.43945429197371044 } }, "acc_norm,none": { "candidate": { "stderr": 0.004940631135803536, "value": 0.5700059749053973 }, "delta_candidate_minus_upstream": 0.002887870942043347, "role": "primary", "untouched_upstream": { "stderr": 0.004944620712318271, "value": 0.5671181039633539 } } }, "sample_count": 10042 }, "mmlu": { "interpretation": "Accuracy over all MMLU subjects; higher is better.", "metrics": { "acc,none": { "candidate": { "stderr": 0.0035976362847934605, "value": 0.2404928072924085 }, "delta_candidate_minus_upstream": 0.0024213075060532663, "role": "primary", "untouched_upstream": { "stderr": 0.0035855541877085647, "value": 0.23807149978635522 } } }, "sample_count": 14042 }, "truthfulqa_mc2": { "interpretation": "Mean normalized probability mass assigned to true answer options; higher is better.", "metrics": { "acc,none": { "candidate": { "stderr": 0.015423514306336993, "value": 0.5378683032141651 }, "delta_candidate_minus_upstream": -0.024131181526524492, "role": "primary", "untouched_upstream": { "stderr": 0.015309998030227038, "value": 0.5619994847406896 } } }, "sample_count": 817 }, "winogrande": { "interpretation": "Multiple-choice accuracy; higher is better.", "metrics": { "acc,none": { "candidate": { "stderr": 0.01375574351374902, "value": 0.6022099447513812 }, "delta_candidate_minus_upstream": -0.0023677979479084232, "role": "primary", "untouched_upstream": { "stderr": 0.013741678387545354, "value": 0.6045777426992897 } } }, "sample_count": 1267 } } }