model: Macaron-V1-Venti source: name: Model Card url: https://huggingface.co/mindlab-research/Macaron-V1-Venti notes: higher_is_better: true missing_values_are_null: false baselines: - Macaron V1 - GLM 5.2 - GPT 5.5 - Claude Opus 4.8 - Gemini 3.1 Pro - Qwen 3.7 Max - Minimax M3 results: - benchmark: ChatBench scores: {Macaron V1: 58.3, GLM 5.2: 54.5, GPT 5.5: 55.5, Claude Opus 4.8: 52.8, Gemini 3.1 Pro: 52.0, Qwen 3.7 Max: 52.5, Minimax M3: 49.1} - benchmark: LivingBench scores: {Macaron V1: 64.0, GLM 5.2: 60.5, GPT 5.5: 61.9, Claude Opus 4.8: 63.8, Gemini 3.1 Pro: 52.1, Qwen 3.7 Max: 56.1, Minimax M3: 57.1} - benchmark: VitaBench scores: {Macaron V1: 60.0, GLM 5.2: 55.8, GPT 5.5: 55.8, Claude Opus 4.8: 56.5, Gemini 3.1 Pro: 55.2, Qwen 3.7 Max: 61.2, Minimax M3: 56.8} - benchmark: PinchBench scores: {Macaron V1: 94.0, GLM 5.2: 88.1, GPT 5.5: 89.0, Claude Opus 4.8: 91.8, Gemini 3.1 Pro: 82.9, Qwen 3.7 Max: 93.4, Minimax M3: 86.1} - benchmark: ClawGym scores: {Macaron V1: 77.7, GLM 5.2: 74.6, GPT 5.5: 82.5, Claude Opus 4.8: 80.5, Gemini 3.1 Pro: 77.5, Qwen 3.7 Max: 75.7, Minimax M3: 76.2} - benchmark: SWE Verified scores: {Macaron V1: 85.6, GLM 5.2: 80.4, GPT 5.5: 82.9, Claude Opus 4.8: 88.6, Gemini 3.1 Pro: 80.6, Qwen 3.7 Max: 80.4, Minimax M3: 80.5} - benchmark: TerminalBench 2.1 scores: {Macaron V1: 87.6, GLM 5.2: 82.7, GPT 5.5: 83.4, Claude Opus 4.8: 78.9, Gemini 3.1 Pro: 70.7, Qwen 3.7 Max: 73.5, Minimax M3: 66.0} - benchmark: DeepSWE scores: {Macaron V1: 58.4, GLM 5.2: 54.9, GPT 5.5: 70.0, Claude Opus 4.8: 58.0, Gemini 3.1 Pro: 10.0, Qwen 3.7 Max: 18.0, Minimax M3: 20.0} - benchmark: SWE Atlas QnA scores: {Macaron V1: 49.5, GLM 5.2: 48.9, GPT 5.5: 45.4, Claude Opus 4.8: 57.3, Gemini 3.1 Pro: 13.5, Qwen 3.7 Max: 22.6, Minimax M3: 37.9} - benchmark: UI4ABench scores: {Macaron V1: 87.8, GLM 5.2: 67.1, GPT 5.5: 72.1, Claude Opus 4.8: 75.9, Gemini 3.1 Pro: 60.3, Qwen 3.7 Max: 62.5, Minimax M3: 63.0}