diff --git "a/README.md" "b/README.md" --- "a/README.md" +++ "b/README.md" @@ -85,7 +85,676 @@ The training data curation process includes cleaning, processing, and modifying ## 5. Evaluations -
| Open weights | Closed weights | |||||||||
|---|---|---|---|---|---|---|---|---|---|---|
Inkling-Small | Qwen3.5-397B-A17B | MiMo V2.5 | Minimax M2.7 | DeepSeek V4 Flash | Nemotron 3 Ultra | Inkling | Claude 4.5 Haiku | Gemini 3.5 Flash-Lite | GPT 5.6 Luna | |
| Model Info | ||||||||||
| AA Index (v4.1)Score | 40.0% | 34.0% | 37.0% | 38.0% | 40.0% | 38.0% | 41.0% | 30.0% | 36.0% | 49.0% |
| Activated Params (B)Score | 12 | 17 | 15 | 10 | 13 | 55 | 41 | – | – | – |
| Total Params (B)Score | 276 | 397 | 310 | 230 | 284 | 550 | 975 | – | – | – |
| Pricing ($/M)Input | 0.3 | 0.39 | 0.14 | 0.25 | 0.14 | 0.5 | 1 | 1 | 0.3 | 0.5 |
| Pricing ($/M)Output | 1.2 | 2.34 | 0.28 | 1 | 0.28 | 2.2 | 4.05 | 5 | 2.5 | 3 |
| Agentic (coding) | ||||||||||
| SWEBench VerifiedScore | 80.2% | 76.4% | 71.0% | 79.9% | 79.0% | 70.7% | 77.6% | 66.6% | 75.0% | 93.0% |
| SWEBench Pro (Public)Score | 55.9% | 50.9% | 56.1% | 56.2% | 52.6% | 46.4% | 54.3% | 39.5% | 54.2% | 62.7% |
| Terminal Bench 2.1Best Harness | 64.69 | 51.3 | 63.7 | 55.4 | 61.8 | 56.4 | 63.8 | 44.2 | 54 | 82.5 |
| SciCodeScore | 48.7% | 42.0% | 43.1% | 47.0% | 44.9% | 39.9% | 46.1% | 43.3% | 40.9% | 50.0% |
| Agentic (general) | ||||||||||
| GDPVal-AA v2Score | 1269 | 962 | 1145 | 1159 | 1189 | 1164 | 1238 | 911 | 1139 | 1530 |
| MCP AtlasPublic | 79.6 | 74.2 | – | 49.4 | 69 | 47.4 | 78.8 | 41.2 | 79.8 | 77 |
| MCP AtlasAll | 79.2 | – | – | – | – | 44.7 | 76 | 40.2 | 76.8 | 75 |
| Tau 3 BankingScore | 15.5% | 13.4% | 6.6% | 8.9% | 22.9% | 13.8% | 23.7% | 9.1% | 16.5% | 24.3% |
| BrowseComp (w/ Ctx)Score | 77.4% | 78.6% | – | 76.3% | 73.2% | 63.0% | 77.1% | – | – | 84.0% |
| Toolathlon-VerifiedScore | 54.4% | – | – | – | – | 34.3% | 45.5% | – | – | – |
| AA-BriefcaseScore | 917 | – | – | – | 833 | 870 | 839 | 612 | – | – |
| Reasoning (general) | ||||||||||
| GPQA DiamondScore | 89.5% | 89.3% | 84.9% | 87.4% | 89.4% | 86.7% | 87.2% | 67.2% | 83.8% | 89.5% |
| HLE (text only)Score | 31.6% | 27.3% | 25.2% | 28.1% | 32.1% | 26.6% | 29.7% | 9.7% | 17.5% | 35.6% |
| HLE (with tools)Score | 47.8% | 48.3% | 40.0% | 40.3% | 45.1% | 37.4% | 46.0% | 17.6% | 42.5% | 48.9% |
| AIME 2026Score | 95.5% | 93.3% | 93.6% | 87.7% | 95.8% | 94.2% | 97.1% | 81.2% | 82.2% | 97.6% |
| HMMT Feb 2026Score | 90.2% | 87.9% | 82.6% | 71.2% | 93.9% | 78.8% | 86.3% | – | – | 98.5% |
| CritPtScore | 8.3% | 1.7% | 3.7% | 0.6% | 7.1% | 3.1% | 5.4% | 0.0% | 0.0% | 20.6% |
| Reasoning (abstract) | ||||||||||
| ARC-AGI-1Score | 84.0% | – | – | – | – | – | 79.5% | 47.7% | – | 87.7% |
| ARC-AGI-2Score | 40.1% | – | – | – | – | ��� | 36.5% | 4.0% | – | 47.6% |
| Factuality | ||||||||||
| SimpleQA VerifiedScore | 20.6% | 26.0% | 16.1% | 13.5% | 34.1% | 32.4% | 43.9% | 5.9% | 44.1% | 41.7% |
| AA OmniscienceScore | -9 | -29.8 | -9.3 | 0.7 | -22.9 | -1 | 2.1 | -4.2 | 6.9 | -11.6 |
| Chat | ||||||||||
| IFBenchScore | 82.2% | 78.8% | 67.1% | 75.7% | 79.2% | 81.4% | 79.8% | 54.3% | 78.6% | 67.3% |
| Global-MMLU-LiteScore | 86.7% | 90.0% | 83.5% | 83.9% | 88.4% | 85.6% | 88.7% | 83.4% | 89.4% | 88.7% |
| Safety | ||||||||||
| StrongREJECT (none)Score | 98.4% | 99.4% | 99.3% | 99.4% | 97.4% | 98.7% | 98.6% | 98.6% | 97.6% | 98.7% |
| FORTRESS (adversarial)Score | 71.6% | 77.3% | 64.8% | 86.3% | 32.0% | 77.6% | 78.0% | 91.3% | 70.7% | 83.8% |
| FORTRESS (benign)Score | 96.9% | 95.4% | 94.6% | 90.1% | 99.2% | 90.5% | 95.9% | 94.1% | 95.5% | 97.8% |
| Vision | ||||||||||
| MMMU Pro (Standard 10)Score | 74.0% | 77.3% | 75.4% | – | – | – | 73.5% | 58.6% | 79.0% | 78.6% |
| Charxiv RQScore | 77.4% | 80.8% | 81.0% | – | – | – | 78.1% | 57.4% | 70.0% | 81.4% |
| Charxiv RQ (with python)Score | 81.3% | – | – | – | – | – | 82.0% | – | – | – |
| Audio | ||||||||||
| Audio MCScore | 54.9% | – | 30.4% | – | – | – | 56.6% | – | 33.6% | – |
| MMAUScore | 77.0% | – | 73.6% | – | – | – | 77.2% | – | 75.2% | – |
| VoiceBenchScore | 90.1% | – | 86.4% | – | – | – | 91.4% | – | 85.9% | – |
| + Open weights + | ++ Closed weights + | +|||||||||
|---|---|---|---|---|---|---|---|---|---|---|
| + Inkling-Small | ++ Qwen3.5 397B-A17B | ++ MiMo V2.5 | ++ Minimax M2.7 | ++ DeepSeek V4 Flash | ++ Nemotron 3 Ultra | ++ Inkling | ++ Claude 4.5 Haiku | ++ Gemini 3.5 Flash-Lite | ++ GPT 5.6 Luna | +|
| Model Info | + + + + + + + + + + +||||||||||
| + AA Index (v4.1) + | +40.0% | +34.0% | +37.0% | +38.0% | +40.0% | +38.0% | +41.0% | +30.0% | +36.0% | +49.0% | +
| + Params (B) (activated / total) + | +12 / 276 | +17 / 397 | +15 / 310 | +10 / 230 | +13 / 284 | +55 / 550 | +41 / 975 | +– | +– | +– | +
| Agentic (coding) | + + + + + + + + + + +||||||||||
| + SWEBench Verified + | +80.2% | +76.4% | +71.0% | +79.9% | +79.0% | +70.7% | +77.6% | +73.3% | +75.0% | +93.0% | +
| + SWEBench Pro (public) + | +55.9% | +50.9% | +56.1% | +56.2% | +52.6% | +46.4% | +54.3% | +39.5% | +54.2% | +62.7% | +
| + Terminal Bench 2.1 (best harness) + | +64.7% | +51.3% | +63.7% | +55.4% | +61.8% | +56.4% | +63.8% | +44.2% | +54.0% | +82.5% | +
| + SciCode + | +48.7% | +42.0% | +43.1% | +47.0% | +44.9% | +39.9% | +46.1% | +43.3% | +40.9% | +50.0% | +
| Agentic (general) | + + + + + + + + + + +||||||||||
| + GDPval-AA v2 + | +1269 | +962 | +1145 | +1159 | +1189 | +1164 | +1238 | +911 | +1139 | +1530 | +
| + MCP Atlas (public / all) + | +79.6/79.2% | +74.2%/– | +– | +49.4%/– | +69.0%/– | +47.4/44.7% | +78.8/76.0% | +41.2/40.2% | +79.8/76.8% | +77.0/75.0% | +
| + Tau 3 Banking + | +15.5% | +13.4% | +6.6% | +8.9% | +22.9% | +13.8% | +23.7% | +9.1% | +16.5% | +24.3% | +
| + BrowseComp (with context management) + | +77.4% | +78.6% | +– | +76.3% | +73.2% | +63.0% | +77.1% | +– | +– | +84.0% | +
| + Toolathlon Verified + | +54.4% | +40.7% | +49.1% | +47.5% | +50.9% | +34.3% | +45.5% | +26.9% | +57.1% | +67.9% | +
| + AA-Briefcase + | +917 | +– | +– | +– | +833 | +870 | +839 | +612 | +– | +– | +
| Reasoning (general) | + + + + + + + + + + +||||||||||
| + GPQA Diamond + | +89.5% | +89.3% | +84.9% | +87.4% | +89.4% | +86.7% | +87.2% | +67.2% | +83.8% | +89.5% | +
| + HLE (text only) + | +31.6% | +27.3% | +25.2% | +28.1% | +32.1% | +26.6% | +29.7% | +9.7% | +17.5% | +35.6% | +
| + HLE (with tools) + | +47.8% | +48.3% | +40.0% | +40.3% | +45.1% | +37.4% | +46.0% | +17.8% | +42.5% | +48.9% | +
| + AIME 2026 + | +95.5% | +93.3% | +93.6% | +87.7% | +95.8% | +94.2% | +97.1% | +85.1% | +82.2% | +97.6% | +
| + HMMT Feb 2026 + | +90.2% | +87.9% | +82.6% | +71.2% | +93.9% | +78.8% | +86.3% | +66.7% | +63.6% | +98.5% | +
| + CritPt + | +8.3% | +1.7% | +3.7% | +0.6% | +7.1% | +3.1% | +5.4% | +0.0% | +0.0% | +20.6% | +
| Reasoning (abstract) | + + + + + + + + + + +||||||||||
| + ARC-AGI-1 + | +84.0% | +– | +– | +– | +– | +– | +79.5% | +47.7% | +– | +87.7% | +
| + ARC-AGI-2 + | +40.1% | +– | +– | +– | +– | +– | +36.5% | +4.0% | +– | +47.6% | +
| Factuality | + + + + + + + + + + +||||||||||
| + SimpleQA Verified + | +20.6% | +26.0% | +16.1% | +13.5% | +34.1% | +32.4% | +43.9% | +5.9% | +44.1% | +41.7% | +
| + AA Omniscience (index) + | +-9.0 | +-29.8 | +-9.3 | +0.7 | +-22.9 | +-1.0 | +2.1 | +-4.2 | +6.9 | +-11.6 | +
| Chat | + + + + + + + + + + +||||||||||
| + IFBench + | +82.2% | +78.8% | +67.1% | +75.7% | +79.2% | +81.4% | +79.8% | +54.3% | +78.6% | +67.3% | +
| + Global-MMLU-Lite + | +86.7% | +90.0% | +83.5% | +83.9% | +88.4% | +85.6% | +88.7% | +83.4% | +89.4% | +88.7% | +
| Safety | + + + + + + + + + + +||||||||||
| + StrongREJECT + | +98.4% | +99.4% | +99.3% | +99.4% | +97.4% | +98.7% | +98.6% | +98.6% | +97.6% | +98.7% | +
| + FORTRESS (adversarial) + | +71.6% | +77.3% | +64.8% | +86.3% | +32.0% | +77.6% | +78.0% | +91.3% | +70.7% | +83.8% | +
| + FORTRESS (benign) + | +96.9% | +95.4% | +94.6% | +90.1% | +99.2% | +90.6% | +95.9% | +94.1% | +95.5% | +97.8% | +
| Vision | + + + + + + + + + + +||||||||||
| + MMMU Pro (Standard 10) + | +74.0% | +77.3% | +75.4% | +– | +– | +– | +73.5% | +58.6% | +79.0% | +78.6% | +
| + Charxiv RQ (original / with python) + | +77.4/81.3% | +80.8%/– | +81.0%/– | +– | +– | +– | +78.1/82.0% | +57.4%/– | +70.0%/– | +81.4%/– | +
| Audio | + + + + + + + + + + +||||||||||
| + Audio MC + | +54.9% | +– | +30.4% | +– | +– | +– | +56.6% | +– | +33.6% | +– | +
| + MMAU + | +77.0% | +– | +73.6% | +– | +– | +– | +77.2% | +– | +75.2% | +– | +
| + VoiceBench + | +90.1% | +– | +86.4% | +– | +– | +– | +91.4% | +– | +85.9% | +– | +