{"kind": "models", "label": "LLM 综合榜", "description": "综合通用智能、人类偏好、知识推理、代码与实时任务表现。", "updated_at": "2026-09-06T10:53:19.604750+08:00", "baseline_updated_at": "2026-08-22T09:00:00+08:00", "methodology_version": "Yeedot AI Rankings 1.0", "benchmarks": [{"key": "aa_index", "label": "Artificial Analysis Intelligence Index", "short_label": "智能", "dimension": "intelligence", "weight": 0.28, "default_weight": 0.28, "scale": {"floor": 20.0, "ceiling": 65.0}, "higher_is_better": true, "source": "Artificial Analysis", "source_url": "https://artificialanalysis.ai/leaderboards/models", "description": "跨知识、数学、代码与 Agentic 工作负载的综合智能指数。"}, {"key": "lmarena", "label": "LMArena Text Arena", "short_label": "偏好", "dimension": "preference", "weight": 0.22, "default_weight": 0.22, "scale": {"floor": 1150.0, "ceiling": 1500.0}, "higher_is_better": true, "source": "LMArena", "source_url": "https://lmarena.ai/leaderboard", "description": "基于匿名成对比较的人类偏好 Elo 分。"}, {"key": "hle", "label": "Humanity's Last Exam", "short_label": "知识", "dimension": "intelligence", "weight": 0.18, "default_weight": 0.18, "scale": {"floor": 10.0, "ceiling": 50.0}, "higher_is_better": true, "source": "Scale Labs", "source_url": "https://labs.scale.com/leaderboard/humanitys_last_exam", "description": "高难度、跨学科专家知识与推理准确率。"}, {"key": "swe_verified", "label": "SWE-bench Verified", "short_label": "代码", "dimension": "coding", "weight": 0.2, "default_weight": 0.2, "scale": {"floor": 30.0, "ceiling": 85.0}, "higher_is_better": true, "source": "SWE-bench", "source_url": "https://www.swebench.com/", "description": "真实 GitHub issue 的端到端软件工程解决率。"}, {"key": "livebench", "label": "LiveBench", "short_label": "实时", "dimension": "freshness", "weight": 0.12, "default_weight": 0.12, "scale": {"floor": 30.0, "ceiling": 85.0}, "higher_is_better": true, "source": "LiveBench", "source_url": "https://livebench.ai/", "description": "通过持续更新题目降低数据污染影响的综合评测。"}], "providers": ["Alibaba", "Anthropic", "DeepSeek", "Google", "Meta", "Mistral AI", "Moonshot AI", "OpenAI", "Z.ai", "xAI"], "entities": [{"id": "claude-opus-5", "name": "Claude Opus 5", "provider": "Anthropic", "summary": "长程推理与知识工作领先，覆盖面均衡。", "open_weights": false, "context": "1M", "tags": ["reasoning", "multimodal"], "accent": "apricot", "metric_provenance": {"livebench": {"identity": "claude-opus-5-max-effort", "name": "claude-opus-5-max-effort", "score": 80.08, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "aa_index": {"identity": "claude-opus-5", "name": "Claude Opus 5 (Adaptive Reasoning, Max Effort)", "score": 54.0539443539342, "scope": "model", "context": "Artificial Analysis public default evaluation; Claude Opus 5 (Adaptive Reasoning, Max Effort)", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "claude-opus-5-max-text", "name": "claude-opus-5-max", "score": 1487.7041429885107, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 87.6, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 80.4, "preference": 96.5, "coding": 92.5, "freshness": 91.0}, "metrics": {"aa_index": {"raw": 54.0539, "normalized": 75.68, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1487.7041, "normalized": 96.49, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 45.1, "normalized": 87.75, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 80.9, "normalized": 92.55, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 80.08, "normalized": 91.05, "weight": 0.12, "source": "LiveBench"}}, "rank": 1}, {"id": "gpt-5-6-sol", "name": "GPT-5.6 Sol", "provider": "OpenAI", "summary": "代码、工具调用与复杂推理表现突出。", "open_weights": false, "context": "400K", "tags": ["reasoning", "tools"], "accent": "mint", "metric_provenance": {"livebench": {"identity": "gpt-5.6-sol-max", "name": "gpt-5.6-sol-max", "score": 81.05, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "aa_index": {"identity": "gpt-5-6-sol", "name": "GPT-5.6 Sol (max)", "score": 51.2551933761681, "scope": "model", "context": "Artificial Analysis public default evaluation; GPT-5.6 Sol (max)", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "gpt-5.6-sol-xhigh-text", "name": "gpt-5.6-sol-xhigh", "score": 1483.1510823374954, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 86.4, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 76.3, "preference": 95.2, "coding": 95.8, "freshness": 92.8}, "metrics": {"aa_index": {"raw": 51.2552, "normalized": 69.46, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1483.1511, "normalized": 95.19, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 44.8, "normalized": 87.0, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 82.7, "normalized": 95.82, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 81.05, "normalized": 92.82, "weight": 0.12, "source": "LiveBench"}}, "rank": 2}, {"id": "muse-spark-1-1", "name": "Muse Spark 1.1", "provider": "Meta", "summary": "开放生态中兼顾代码与通用能力的高效模型。", "open_weights": true, "context": "256K", "tags": ["open weights", "coding"], "accent": "coral", "metric_provenance": {"livebench": {"identity": "muse-spark-1.1-xhigh", "name": "muse-spark-1.1-xhigh", "score": 75.3, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "lmarena": {"identity": "super-nova-ext-3tam-text", "name": "muse-spark-1.1", "score": 1492.058078698562, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 84.4, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 74.6, "preference": 97.7, "coding": 93.5, "freshness": 82.4}, "metrics": {"aa_index": {"raw": 53, "normalized": 73.33, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1492.0581, "normalized": 97.73, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 40.6, "normalized": 76.5, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 81.4, "normalized": 93.45, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 75.3, "normalized": 82.36, "weight": 0.12, "source": "LiveBench"}}, "rank": 3}, {"id": "kimi-k3", "name": "Kimi K3", "provider": "Moonshot AI", "summary": "开放权重前沿模型，长上下文与 Agentic 能力出色。", "open_weights": true, "context": "256K", "tags": ["open weights", "reasoning"], "accent": "blue", "metric_provenance": {"livebench": {"identity": "kimi-k3", "name": "kimi-k3", "score": 79.19, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "aa_index": {"identity": "kimi-k3", "name": "Kimi K3 (max)", "score": 50.2337145305853, "scope": "model", "context": "Artificial Analysis public default evaluation; Kimi K3 (max)", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "kimi-k3-v3-text", "name": "kimi-k3-max", "score": 1488.665886033902, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 83.0, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 72.8, "preference": 96.8, "coding": 87.5, "freshness": 89.4}, "metrics": {"aa_index": {"raw": 50.2337, "normalized": 67.19, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1488.6659, "normalized": 96.76, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 42.6, "normalized": 81.5, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 78.1, "normalized": 87.45, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 79.19, "normalized": 89.44, "weight": 0.12, "source": "LiveBench"}}, "rank": 4}, {"id": "glm-5-2", "name": "GLM-5.2", "provider": "Z.ai", "summary": "强代码表现与较好的 Agentic 可控性。", "open_weights": true, "context": "200K", "tags": ["open weights", "coding"], "accent": "cyan", "metric_provenance": {"livebench": {"identity": "glm-5.2", "name": "glm-5.2", "score": 73.16, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "lmarena": {"identity": "glm-5.2-text", "name": "glm-5.2-max", "score": 1471.71832610685, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 81.2, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 72.9, "preference": 91.9, "coding": 90.2, "freshness": 78.5}, "metrics": {"aa_index": {"raw": 53, "normalized": 73.33, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1471.7183, "normalized": 91.92, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 38.9, "normalized": 72.25, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 79.6, "normalized": 90.18, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 73.16, "normalized": 78.47, "weight": 0.12, "source": "LiveBench"}}, "rank": 5}, {"id": "grok-4-5", "name": "Grok 4.5", "provider": "xAI", "summary": "实时知识与推理速度具备竞争力。", "open_weights": false, "context": "256K", "tags": ["reasoning", "realtime"], "accent": "gray", "metric_provenance": {"livebench": {"identity": "grok-4.5", "name": "grok-4.5", "score": 75.77, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "aa_index": {"identity": "grok-4-5", "name": "Grok 4.5 (high)", "score": 45.4782280369133, "scope": "model", "context": "Artificial Analysis public default evaluation; Grok 4.5 (high)", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "grok-4.5-text", "name": "grok-4.5", "score": 1471.0801583876907, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 75.7, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 65.5, "preference": 91.7, "coding": 76.9, "freshness": 83.2}, "metrics": {"aa_index": {"raw": 45.4782, "normalized": 56.62, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1471.0802, "normalized": 91.74, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 41.7, "normalized": 79.25, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 72.3, "normalized": 76.91, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 75.77, "normalized": 83.22, "weight": 0.12, "source": "LiveBench"}}, "rank": 6}, {"id": "gemini-3-1-pro", "name": "Gemini 3.1 Pro", "provider": "Google", "summary": "多模态理解与科学推理能力强。", "open_weights": false, "context": "2M", "tags": ["multimodal", "reasoning"], "accent": "violet", "metric_provenance": {"livebench": {"identity": "gemini-3.1-pro-preview-high", "name": "gemini-3.1-pro-preview-high", "score": 76.95, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "hle": {"identity": "gemini-3.1-pro-preview (thinking high)", "name": "gemini-3.1-pro-preview (thinking high)", "score": 46.44, "scope": "model", "context": "Scale evaluation; ", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://labs.scale.com/leaderboard/humanitys_last_exam"}, "aa_index": {"identity": "gemini-3-1-pro-preview", "name": "Gemini 3.1 Pro Preview", "score": 36.6542313912925, "scope": "model", "context": "Artificial Analysis public default evaluation; Gemini 3.1 Pro Preview", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "gemini-3.1-pro-preview", "name": "gemini-3.1-pro-preview", "score": 1486.7298381462967, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 74.4, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 58.2, "preference": 96.2, "coding": 80.9, "freshness": 85.4}, "metrics": {"aa_index": {"raw": 36.6542, "normalized": 37.01, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1486.7298, "normalized": 96.21, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 46.44, "normalized": 91.1, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 74.5, "normalized": 80.91, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 76.95, "normalized": 85.36, "weight": 0.12, "source": "LiveBench"}}, "rank": 7}, {"id": "deepseek-v4-flash", "name": "DeepSeek V4 Flash", "provider": "DeepSeek", "summary": "开放权重、成本效率与代码能力平衡。", "open_weights": true, "context": "128K", "tags": ["open weights", "efficient"], "accent": "indigo", "metric_provenance": {"livebench": {"identity": "deepseek-v4-flash-0731", "name": "deepseek-v4-flash-0731", "score": 74.17, "scope": "model", "context": "LiveBench 2026-06-25; mean of category means", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://livebench.ai/"}, "aa_index": {"identity": "deepseek-v4-flash", "name": "DeepSeek V4 Flash 0731 (Reasoning, Max Effort)", "score": 40.8390714993678, "scope": "model", "context": "Artificial Analysis public default evaluation; DeepSeek V4 Flash 0731 (Reasoning, Max Effort)", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "deepseek-v4-ch3-text", "name": "deepseek-v4-flash", "score": 1435.749787126345, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 70.1, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 55.4, "preference": 81.6, "coding": 85.1, "freshness": 80.3}, "metrics": {"aa_index": {"raw": 40.8391, "normalized": 46.31, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1435.7498, "normalized": 81.64, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 37.8, "normalized": 69.5, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 76.8, "normalized": 85.09, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 74.17, "normalized": 80.31, "weight": 0.12, "source": "LiveBench"}}, "rank": 8}, {"id": "qwen-3-8", "name": "Qwen 3.8 72B", "provider": "Alibaba", "summary": "多语言与开放部署场景表现稳定。", "open_weights": true, "context": "128K", "tags": ["open weights", "multilingual"], "accent": "amber", "overall": 69.1, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 62.4, "preference": 70.6, "coding": 78.5, "freshness": 76.2}, "metrics": {"aa_index": {"raw": 48, "normalized": 62.22, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1397, "normalized": 70.57, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 35.1, "normalized": 62.75, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 73.2, "normalized": 78.55, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 71.9, "normalized": 76.18, "weight": 0.12, "source": "LiveBench"}}, "rank": 9}, {"id": "mistral-large-3", "name": "Mistral Large 3", "provider": "Mistral AI", "summary": "欧洲生态中的高效通用模型。", "open_weights": false, "context": "128K", "tags": ["multilingual", "efficient"], "accent": "orange", "metric_provenance": {"aa_index": {"identity": "mistral-large-3", "name": "Mistral Large 3", "score": 11.1335262865373, "scope": "model", "context": "Artificial Analysis public default evaluation; Mistral Large 3", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://artificialanalysis.ai/leaderboards/models"}, "lmarena": {"identity": "mistral-large-3", "name": "mistral-large-3", "score": 1413.6667924340622, "scope": "model", "context": "Text overall; styleControl=True", "checked_at": "2026-09-06T16:30:00.000236+08:00", "source_url": "https://lmarena.ai/leaderboard"}}, "overall": 50.3, "coverage": 100, "confidence": "高", "dimensions": {"intelligence": 23.2, "preference": 75.3, "coding": 71.6, "freshness": 73.1}, "metrics": {"aa_index": {"raw": 11.1335, "normalized": 0.0, "weight": 0.28, "source": "Artificial Analysis"}, "lmarena": {"raw": 1413.6668, "normalized": 75.33, "weight": 0.22, "source": "LMArena"}, "hle": {"raw": 33.7, "normalized": 59.25, "weight": 0.18, "source": "Scale Labs"}, "swe_verified": {"raw": 69.4, "normalized": 71.64, "weight": 0.2, "source": "SWE-bench"}, "livebench": {"raw": 70.2, "normalized": 73.09, "weight": 0.12, "source": "LiveBench"}}, "rank": 10}], "count": 10}