{ "pdf_sha256": "2cfb6d1b55af14955b4634f3033bbf4f11740e9c7c1df770e6b748d370b67da3", "n_pages": 25, "benchmarks": [ { "name": "DistilBERT", "d": 14, "m": 1440, "kind": "deep learning (text)" }, { "name": "Estate", "d": 15, "m": 1722, "kind": "tabular" }, { "name": "ViT16", "d": 16, "m": 1705, "kind": "deep learning (vision)" }, { "name": "Cancer", "d": 30, "m": 2281, "kind": "tabular" }, { "name": "IL60", "d": 60, "m": 5521, "kind": "tabular" }, { "name": "CG60", "d": 60, "m": 5521, "kind": "tabular" }, { "name": "NHANES", "d": 79, "m": 5864, "kind": "tabular" }, { "name": "Crime", "d": 101, "m": 11126, "kind": "tabular" } ], "claim4": { "assertion_n_estimators": 8, "actual_n_estimator_rows": 11, "estimator_rows": [ "MSR", "SV ARM", "PermutationSampling", "LeverageSHAP", "PolySHAP-3", "RegressionMSR", "Proxy", "FFD-RD", "FFD-RD-Corrected", "FourierSHAP", "OddSHAP" ], "n_estimators_matches": false, "assertion_n_benchmarks": 8, "actual_n_benchmarks": 8, "n_benchmarks_matches": true, "assertion_oddshap_rank": 1.5, "actual_oddshap_rank": 1.5, "oddshap_rank_matches": true, "oddshap_is_lowest_rank": true, "assertion_regressionmsr_rank": 2.25, "actual_regressionmsr_rank": 2.62, "regressionmsr_rank_matches": false, "second_best_estimator": "RegressionMSR", "all_ranks": { "MSR": 7.75, "SV ARM": 6.75, "PermutationSampling": 5.62, "LeverageSHAP": 3.25, "PolySHAP-3": 3.25, "RegressionMSR": 2.62, "Proxy": 5.0, "FFD-RD": 5.75, "FFD-RD-Corrected": 8.5, "FourierSHAP": 9.12, "OddSHAP": 1.5 }, "verdict": "FALSIFIED: OddSHAP's rank (1.50) and its first-place finish are correct, and there are indeed 8 benchmarks, but the table lists 11 estimator rows rather than 8, and RegressionMSR's average rank is 2.62, not 2.25." }, "claim5": { "rows": [ { "benchmark": "DistilBERT", "d": 14, "kind": "deep learning (text)", "leverageshap_mse": 7.7e-05, "oddshap_mse": 4.6e-05, "ratio": 1.673913043478261 }, { "benchmark": "Estate", "d": 15, "kind": "tabular", "leverageshap_mse": 0.00034, "oddshap_mse": 5.1e-07, "ratio": 666.6666666666667 }, { "benchmark": "ViT16", "d": 16, "kind": "deep learning (vision)", "leverageshap_mse": 3.5e-05, "oddshap_mse": 1.3e-05, "ratio": 2.692307692307692 }, { "benchmark": "Cancer", "d": 30, "kind": "tabular", "leverageshap_mse": 3.2e-05, "oddshap_mse": 4.2e-06, "ratio": 7.6190476190476195 }, { "benchmark": "IL60", "d": 60, "kind": "tabular", "leverageshap_mse": 2.6e-05, "oddshap_mse": 1.6e-06, "ratio": 16.25 }, { "benchmark": "CG60", "d": 60, "kind": "tabular", "leverageshap_mse": 2.5e-05, "oddshap_mse": 6.2e-06, "ratio": 4.032258064516129 }, { "benchmark": "NHANES", "d": 79, "kind": "tabular", "leverageshap_mse": 0.00062, "oddshap_mse": 5.8e-05, "ratio": 10.689655172413794 }, { "benchmark": "Crime", "d": 101, "kind": "tabular", "leverageshap_mse": 0.75, "oddshap_mse": 0.13, "ratio": 5.769230769230769 } ], "claimed_interval": [ 6.0, 62.0 ], "tabular_only": { "n": 6, "min_ratio": 4.032258064516129, "max_ratio": 666.6666666666667, "ratios": { "Estate": 666.6666666666667, "Cancer": 7.6190476190476195, "IL60": 16.25, "CG60": 4.032258064516129, "NHANES": 10.689655172413794, "Crime": 5.769230769230769 }, "n_inside_interval": 3, "n_below_6": 2, "n_above_62": 1 }, "all_benchmarks": { "n": 8, "min_ratio": 1.673913043478261, "max_ratio": 666.6666666666667 }, "verdict": "FALSIFIED: on the paper's own tabular benchmarks the LeverageSHAP/OddSHAP MSE ratio spans 4.03x to 666.7x, which falls outside the claimed 6-62x band at BOTH ends." }, "claim6": { "complete_rows_used": [ "MSR", "SV ARM", "PermutationSampling", "LeverageSHAP", "RegressionMSR", "Proxy", "FourierSHAP", "OddSHAP" ], "partial_rows_excluded": [ "PolySHAP-3", "FFD-RD", "FFD-RD-Corrected" ], "exclusion_reason": "these rows have blank cells in Table 1 and the linearised PDF text does not determine which benchmarks their numbers belong to", "rows": [ { "benchmark": "DistilBERT", "d": 14, "best_baseline": "RegressionMSR", "best_baseline_mse": 3.1e-05, "oddshap_mse": 4.6e-05, "advantage": 0.6739130434782609 }, { "benchmark": "Estate", "d": 15, "best_baseline": "RegressionMSR", "best_baseline_mse": 1.3e-05, "oddshap_mse": 5.1e-07, "advantage": 25.49019607843137 }, { "benchmark": "ViT16", "d": 16, "best_baseline": "RegressionMSR", "best_baseline_mse": 1e-05, "oddshap_mse": 1.3e-05, "advantage": 0.7692307692307694 }, { "benchmark": "Cancer", "d": 30, "best_baseline": "LeverageSHAP", "best_baseline_mse": 3.2e-05, "oddshap_mse": 4.2e-06, "advantage": 7.6190476190476195 }, { "benchmark": "IL60", "d": 60, "best_baseline": "RegressionMSR", "best_baseline_mse": 9.7e-06, "oddshap_mse": 1.6e-06, "advantage": 6.062500000000001 }, { "benchmark": "CG60", "d": 60, "best_baseline": "LeverageSHAP", "best_baseline_mse": 2.5e-05, "oddshap_mse": 6.2e-06, "advantage": 4.032258064516129 }, { "benchmark": "NHANES", "d": 79, "best_baseline": "RegressionMSR", "best_baseline_mse": 0.00024, "oddshap_mse": 5.8e-05, "advantage": 4.137931034482759 }, { "benchmark": "Crime", "d": 101, "best_baseline": "RegressionMSR", "best_baseline_mse": 0.56, "oddshap_mse": 0.13, "advantage": 4.307692307692308 } ], "max_advantage_benchmark": "Estate", "max_advantage_d": 15, "max_advantage": 25.49019607843137, "low_dim_d_lt_30": { "n": 3, "max_advantage": 25.49019607843137, "median_advantage": 0.7692307692307694 }, "high_dim_d_ge_30": { "n": 5, "max_advantage": 7.6190476190476195, "median_advantage": 4.307692307692308 }, "n_benchmarks_where_oddshap_loses": 2, "n_wins_d_lt_30": 1, "n_wins_d_ge_30": 5, "reading_maximum": { "argmax_d": 15, "argmax_advantage": 25.49019607843137, "supports_claim": false }, "reading_median": { "median_lt_30": 0.7692307692307694, "median_ge_30": 4.307692307692308, "supports_claim": true }, "reading_consistency": { "wins_lt_30": "1/3", "wins_ge_30": "5/5", "supports_claim": true }, "verdict": "MIXED, and we decline to call it falsified. Table 1 at m~100d answers the claim differently depending on what 'largest advantage' means. Under a single-maximum reading the claim FAILS: the biggest advantage is 25.5x at d=15 (Estate), below 30. Under a median or consistency reading the claim HOLDS: the median advantage is 4.31x at d>=30 versus 0.77x below 30, and OddSHAP beats every complete-row baseline on 5/5 of the d>=30 benchmarks but only 1/3 of the smaller ones. The d<30 group is bimodal (one 25.5x win, two losses) while the d>=30 group is uniformly 4-8x, so the maximum is an outlier and the typical behaviour supports the paper." } }