{
  "edition": {
    "checked_at": "2026-09-15T06:34:08+00:00",
    "summary": "本版新增“各家最新模型”与真正多名次榜单：最新卡片只采用官方可用性证据，能力榜仅引用同一份 Artificial Analysis 指标，不把最新发布等同于最强。当前可核实覆盖集中在 Anthropic、DeepSeek、Qwen；OpenAI、Google、xAI、Meta、Mistral 的最新型号待下一轮官方页面核查。",
    "recommendations": [
      {
        "model": "Claude Fable 5.1",
        "category": "overall",
        "reason": "重质量、预算宽裕可优先试：AA 在 9 月 1 日报告其 max 档 Intelligence Index 为 66。",
        "caveat": "这是该报告覆盖范围内的推荐，不保证所有新模型中的第一；max 档费用较高。",
        "source_date": "2026-09-01",
        "release": {
          "source": "anthropic-fable",
          "quote": "Claude Fable 5.1 and Claude Mythos 5.1 are the same model, but with different levels of safeguards. Fable 5.1 is generally available, while Mythos 5.1 is available only through our trusted access programs; its safeguards are specifically designed to support work in cybersecurity and the life sciences. Alongside its increased capabilities, Fable 5.1 takes important steps towards addressing the feedback we’ve received "
        },
        "evaluation": {
          "source": "aa-fable",
          "quote": "We supported Anthropic with pre-release evaluation of Claude Fable 5.1. At max effort it scores 66 on the Artificial Analysis Intelligence Index, the highest score we have measured, ahead of Claude Opus 5 (max, 63), Claude Fable 5 (max, 62), GPT-5.6 Sol (max, 61) and Grok 4.6 (high, 61). We evaluated the model with Anthropic's 'default' server-side fallback, which routes safety-flagged requests to Claude Opus 4.8 or Claude Opus 5; fallback served ~4% of output tokens across the Intelligence Index. Key takeaways: ➤ Frontier intelligence with improvements across benchmarks: Fable 5.1 gains +4 points on the Intelligence Index over Fable 5. On HLE, Fable 5.1 scores 59.1%, ahead of the previous best of 55.5% from Claude Fable 5. It posts the narrowly highest scores we've seen on Terminal-Bench v2.1 (91.4%) and SciCode (62.0%), and on 𝜏³-Banking it gains 9 points over Fable 5 ➤ 75% cache read price cut, but Fable 5.1 still costs more per task: Anthropic has cut the cache read price from $1 to $0.25 per 1M cached input tokens, with standard pricing unchanged at $10/$50 per 1M input/output tokens. Fable 5.1 (max) costs $3.76 per Intelligenc"
        }
      },
      {
        "model": "Claude Fable 5.1",
        "category": "coding",
        "reason": "适合复杂编程任务：AA 报告 Terminal-Bench v2.1 与 SciCode 表现突出。",
        "caveat": "测试使用特定配置；终端代理、前端界面与实际仓库修复不是同一任务。",
        "source_date": "2026-09-01",
        "release": {
          "source": "anthropic-fable",
          "quote": "Claude Fable 5.1 and Claude Mythos 5.1 are the same model, but with different levels of safeguards. Fable 5.1 is generally available, while Mythos 5.1 is available only through our trusted access programs; its safeguards are specifically designed to support work in cybersecurity and the life sciences. Alongside its increased capabilities, Fable 5.1 takes important steps towards addressing the feedback we’ve received "
        },
        "evaluation": {
          "source": "aa-fable",
          "quote": "We supported Anthropic with pre-release evaluation of Claude Fable 5.1. At max effort it scores 66 on the Artificial Analysis Intelligence Index, the highest score we have measured, ahead of Claude Opus 5 (max, 63), Claude Fable 5 (max, 62), GPT-5.6 Sol (max, 61) and Grok 4.6 (high, 61). We evaluated the model with Anthropic's 'default' server-side fallback, which routes safety-flagged requests to Claude Opus 4.8 or Claude Opus 5; fallback served ~4% of output tokens across the Intelligence Index. Key takeaways: ➤ Frontier intelligence with improvements across benchmarks: Fable 5.1 gains +4 points on the Intelligence Index over Fable 5. On HLE, Fable 5.1 scores 59.1%, ahead of the previous best of 55.5% from Claude Fable 5. It posts the narrowly highest scores we've seen on Terminal-Bench v2.1 (91.4%) and SciCode (62.0%), and on 𝜏³-Banking it gains 9 points over Fable 5 ➤ 75% cache read price cut, but Fable 5.1 still costs more per task: Anthropic has cut the cache read price from $1 to $0.25 per 1M cached input tokens, with standard pricing unchanged at $10/$50 per 1M input/output tokens. Fable 5.1 (max) costs $3.76 per Intelligenc"
        }
      },
      {
        "model": "Claude Fable 5.1",
        "category": "reasoning",
        "reason": "需要复杂知识与推理时可试：AA 的 HLE 评测显示相对上一版有提升。",
        "caveat": "模型仍可能产生幻觉；专业结论需复核，不能将 HLE 当成所有推理任务的保证。",
        "source_date": "2026-09-01",
        "release": {
          "source": "anthropic-fable",
          "quote": "Claude Fable 5.1 and Claude Mythos 5.1 are the same model, but with different levels of safeguards. Fable 5.1 is generally available, while Mythos 5.1 is available only through our trusted access programs; its safeguards are specifically designed to support work in cybersecurity and the life sciences. Alongside its increased capabilities, Fable 5.1 takes important steps towards addressing the feedback we’ve received "
        },
        "evaluation": {
          "source": "aa-fable",
          "quote": "We supported Anthropic with pre-release evaluation of Claude Fable 5.1. At max effort it scores 66 on the Artificial Analysis Intelligence Index, the highest score we have measured, ahead of Claude Opus 5 (max, 63), Claude Fable 5 (max, 62), GPT-5.6 Sol (max, 61) and Grok 4.6 (high, 61). We evaluated the model with Anthropic's 'default' server-side fallback, which routes safety-flagged requests to Claude Opus 4.8 or Claude Opus 5; fallback served ~4% of output tokens across the Intelligence Index. Key takeaways: ➤ Frontier intelligence with improvements across benchmarks: Fable 5.1 gains +4 points on the Intelligence Index over Fable 5. On HLE, Fable 5.1 scores 59.1%, ahead of the previous best of 55.5% from Claude Fable 5. It posts the narrowly highest scores we've seen on Terminal-Bench v2.1 (91.4%) and SciCode (62.0%), and on 𝜏³-Banking it gains 9 points over Fable 5 ➤ 75% cache read price cut, but Fable 5.1 still costs more per task: Anthropic has cut the cache read price from $1 to $0.25 per 1M cached input tokens, with standard pricing unchanged at $10/$50 per 1M input/output tokens. Fable 5.1 (max) costs $3.76 per Intelligenc"
        }
      },
      {
        "model": "Claude Opus 5",
        "category": "value",
        "reason": "在本次核查的高端模型之间偏重成本：AA 报告 max 档每个指数任务约 $2.34，低于 Fable 5.1 max 的 $3.76。",
        "caveat": "只比较这份评测中的高端模型，不是全市场最便宜；真实成本取决于输出长度、缓存和任务。",
        "source_date": "2026-09-01",
        "release": {
          "source": "anthropic-opus",
          "quote": "Claude Opus 5 is available today. It’s a thoughtful and proactive model that comes close to the frontier intelligence of Claude Fable 5 at half the price. On coding and knowledge work evaluations like Frontier-Bench and GDPval-AA , Opus 5 is the new state-of-the-art, though it remains behind Mythos "
        },
        "evaluation": {
          "source": "aa-fable",
          "quote": "At xhigh effort Fable 5.1 scores 65 at $2.72 per task, $1.04 less than max, but still above Claude Opus 5 (max, 63) at $2.34 ➤ Claude Fable 5.1 holds the upper end of the Intelligence vs. Output Tokens per Task Pareto frontier: every model variant scoring higher than GPT-5.6 Sol (medium) on the Intelligence Index is matched or beaten by a Fable 5.1 effort level on bot"
        }
      }
    ],
    "rumors": [],
    "pending": [],
    "history": [],
    "latest_models": [
      {
        "vendor": "Anthropic",
        "model": "Claude Fable 5.1",
        "release_date": "2026-08-02",
        "status": "已发布 · 可用",
        "highlight": "官方称其面向编程、知识工作和长程问题解决；这里只把它列为最新发布，不等于自动获得所有榜单第一。",
        "official": {
          "source": "anthropic-fable",
          "quote": "Claude Fable 5.1 and Claude Mythos 5.1 are the same model, but with different levels of safeguards. Fable 5.1 is generally available, while Mythos 5.1 is available only through our trusted access programs; its safeguards are specifically designed to support work in cybersecurity and the life sciences."
        }
      },
      {
        "vendor": "Anthropic",
        "model": "Claude Opus 5",
        "release_date": "2026-07-24",
        "status": "已发布 · 可用",
        "highlight": "官方发布页称其在 Claude 平台、Claude Code 等渠道可用；能力排名另见同口径评测榜。",
        "official": {
          "source": "anthropic-opus",
          "quote": "Claude Opus 5 is available today. It’s a thoughtful and proactive model that comes close to the frontier intelligence of Claude Fable 5 at half the price."
        }
      },
      {
        "vendor": "DeepSeek",
        "model": "DeepSeek-V4.1-Flash",
        "release_date": "2026-09-14",
        "status": "API 可用",
        "highlight": "官方 API 文档说明旧版名称的请求由 DeepSeek-V4.1-Flash 提供服务；这是接口文档核实，不是独立能力排名。",
        "official": {
          "source": "deepseek-api",
          "quote": "The legacy names deepseek-v4-flash and deepseek-v4-flash-vision-exp are still accepted, but the corresponding models have been retired, their requests are served by the DeepSeek-V4.1-Flash model and billed at the Flash price."
        }
      },
      {
        "vendor": "Qwen",
        "model": "Qwen3Guard",
        "release_date": "2025-09-23",
        "status": "已发布 · 安全模型",
        "highlight": "Qwen 官方博客介绍的安全护栏模型；它不是通用聊天模型，因此不与通用模型直接争夺综合榜名次。",
        "official": {
          "source": "qwen-blog",
          "quote": "We are excited to introduce Qwen3Guard, the first safety guardrail model in the Qwen family. Built upon the powerful Qwen3 foundation models and fine-tuned specifically for safety classification, Qwen3Guard ensures responsible AI interactions by delivering precise safety detection for both prompts and responses."
        }
      }
    ],
    "rankings": [
      {
        "id": "aa-intelligence-2026-09-01",
        "title": "Artificial Analysis Intelligence Index",
        "dimension": "综合智能指数（max/high effort，数值越高越前）",
        "source_date": "2026-09-01",
        "evidence": {
          "source": "aa-fable",
          "quote": "We supported Anthropic with pre-release evaluation of Claude Fable 5.1. At max effort it scores 66 on the Artificial Analysis Intelligence Index, the highest score we have measured, ahead of Claude Opus 5 (max, 63), Claude Fable 5 (max, 62), GPT-5.6 Sol (max, 61) and Grok 4.6 (high, 61). We evaluated the model with Anthropic's 'default' server-side fallback, which routes safety-flagged requests to Claude Opus 4.8 or Claude Opus 5; fallback served ~4% of output tokens across the Intelligence Index. Key takeaways: ➤ Frontier intelligence with improvements across benchmarks: Fable 5.1 gains +4 points on the Intelligence Index over Fable 5. On HLE, Fable 5.1 scores 59.1%, ahead of the previous best of 55.5% from Claude Fable 5. It posts the narrowly highest scores we've seen on Terminal-Bench v2.1 (91.4%) and SciCode (62.0%), and on 𝜏³-Banking it gains 9 points over Fable 5 ➤ 75% cache read price cut, but Fable 5.1 still costs more per task: Anthropic has cut the cache read price from $1 to $0.25 per 1M cached input tokens, with standard "
        },
        "entries": [
          {
            "rank": 1,
            "model": "Claude Fable 5.1",
            "vendor": "Anthropic",
            "score": 66,
            "unit": "分"
          },
          {
            "rank": 2,
            "model": "Claude Opus 5",
            "vendor": "Anthropic",
            "score": 63,
            "unit": "分"
          },
          {
            "rank": 3,
            "model": "Claude Fable 5",
            "vendor": "Anthropic",
            "score": 62,
            "unit": "分"
          },
          {
            "rank": 4,
            "model": "GPT-5.6 Sol",
            "vendor": "OpenAI",
            "score": 61,
            "unit": "分"
          },
          {
            "rank": 5,
            "model": "Grok 4.6",
            "vendor": "xAI",
            "score": 61,
            "unit": "分"
          }
        ]
      }
    ]
  },
  "sources": {
    "aa-fable": {
      "url": "https://artificialanalysis.ai/articles/claude-fable-5-1",
      "title": "artificialanalysis.ai · articles/claude-fable-5-1",
      "kind": "independent",
      "fetched_at": "2026-09-15T06:31:53+00:00",
      "dates": [
        "2026-09-01",
        "2026-09-07",
        "2026-09-09"
      ]
    },
    "anthropic-fable": {
      "url": "https://www.anthropic.com/claude-fable-and-mythos-5-1",
      "title": "www.anthropic.com · claude-fable-and-mythos-5-1",
      "kind": "official",
      "fetched_at": "2026-09-15T06:31:58+00:00",
      "dates": [
        "2026-08-02"
      ]
    },
    "anthropic-opus": {
      "url": "https://www.anthropic.com/news/claude-opus-5",
      "title": "www.anthropic.com · news/claude-opus-5",
      "kind": "official",
      "fetched_at": "2026-09-15T06:32:00+00:00",
      "dates": [
        "2026-07-24"
      ]
    },
    "deepseek-api": {
      "url": "https://api-docs.deepseek.com/news/",
      "title": "DeepSeek API Docs · News / model names",
      "kind": "official",
      "fetched_at": "2026-09-15T07:00:00+00:00",
      "dates": [
        "2026-09-14"
      ]
    },
    "qwen-blog": {
      "url": "https://qwenlm.github.io/blog/",
      "title": "Qwen Blog · Qwen3Guard",
      "kind": "official",
      "fetched_at": "2026-09-15T07:00:00+00:00",
      "dates": [
        "2025-09-23"
      ]
    }
  }
}