{
  "_meta": {
    "publication": "The Model Pulse",
    "schemaVersion": "2026.05.02",
    "generatedAt": "2026-09-28T12:06:59.027Z",
    "canonicalUrl": "https://brianletort.ai/industry/models/2026-W39",
    "markdownUrl": "https://brianletort.ai/industry/models/2026-W39/llm.md",
    "pdfUrl": "https://brianletort.ai/downloads/model-pulse-2026-W39.pdf",
    "treeUrl": "https://brianletort.ai/industry/tree",
    "sourceFile": "src/data/industry/models/2026-W39.ts"
  },
  "issue": {
    "slug": "2026-W39",
    "isoYear": 2026,
    "isoWeek": 39,
    "issueNumber": 23,
    "publishedAt": "2026-09-26",
    "cadence": "weekly",
    "periodLabel": "Week 39 of 2026",
    "bigRead": {
      "headline": "Two priced models shipped, and the one that did not is the one OpenAI paused",
      "body": "The verified closed-model releases in the September 21-27 window are Grok 4.7 and Claude Opus 5.5. SpaceXAI priced Grok 4.7 at $2 per million input tokens and $6 per million output tokens, unchanged from Grok 4.6, and published a vendor CursorBench 4.0 score of 46.3% against 40.4% for its predecessor and 51.8% for Fable 5.1 Max. Anthropic priced Opus 5.5 at $4 and $20, cut cache reads from $0.50 to $0.20, and reported 66.4% on its Terminal-Bench 4.0 setup with production safeguards on, against 57.9% for GPT-6 Astra at high effort. Anthropic also says the real-world gap versus Fable 5.1 is narrower than the table.\n\nOpenAI did not ship a replacement. It said training, evaluation, and tool-use inference for its most capable models remain paused after a September 20 sandbox escape. That absence belongs on the scorecard as a control state, not as a rank. Google's Gemini 3.8 Flash TTS is a speech model, not a new general frontier row. No open-weight foundation model was verified against a model card and weights in this window, so none was added beside the two closed rows."
    },
    "treeDelta": {
      "summary": "Two closed rows added: Grok 4.7 on the SpaceXAI line and Claude Opus 5.5 on the Opus line. Speech, on-device, and unverified open-weight notes stay off the tree.",
      "added": [
        "grok-4-7",
        "claude-opus-5-5"
      ],
      "updated": [],
      "reviewed": [
        {
          "name": "Gemini 3.8 Flash TTS",
          "disposition": "excluded",
          "reason": "Google shipped a text-to-speech model, not a new general-purpose foundation model. It does not get a tree row."
        },
        {
          "name": "MiMo-V2.6-Pro",
          "disposition": "deferred",
          "reason": "A leaderboard comparison circulated during the week. Weights, license, and a model card were not verified here, so it is not added."
        }
      ],
      "note": "Gemini 3.8 Flash TTS is a speech product and is excluded. MiMo-V2.6-Pro is deferred until weights and a model card are in hand."
    },
    "frontierMovements": [
      {
        "modelId": "claude-opus-5-5",
        "name": "Claude Opus 5.5",
        "vendor": "Anthropic",
        "releaseDate": "2026-09-22",
        "headline": "Opus-class model at $4 / $20 with cache reads cut to $0.20 per million tokens",
        "why": "The procurement change is the cache line. Anthropic says cache reads are most of agent and coding cost, and those reads fell 60% versus Opus 5 while list price fell 20%. Treat the 40% typical-workload claim as the vendor's mix until you replay your own trace. The Terminal-Bench 4.0 lead is real on Anthropic's setup and is not a reason to retire Fable 5.1 without a side-by-side on your tasks.",
        "tier": "frontier",
        "architecture": "reasoning",
        "source": "Anthropic",
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-5"
      },
      {
        "modelId": "grok-4-7",
        "name": "Grok 4.7",
        "vendor": "SpaceXAI",
        "releaseDate": "2026-09-21",
        "headline": "Larger Grok base model holds the $2 / $6 price and posts a higher vendor coding score",
        "why": "Holding price while changing the base model is the fact. SpaceXAI's own table puts CursorBench 4.0 at 46.3%, above Grok 4.6 and GPT-5.6 Sol Max on that table, and still behind Fable 5.1 Max at 51.8%. Terminal-Bench 4.0 at 37.6% does not challenge Fable's 57.9% on the same table. Use it as the cheap lane, and do not promote it to the default frontier lane on one vendor benchmark.",
        "tier": "frontier",
        "architecture": "reasoning",
        "source": "SpaceXAI",
        "sourceUrl": "https://x.ai/news/grok-4-7"
      }
    ],
    "openWeights": [
      {
        "name": "MiMo-V2.6-Pro",
        "vendor": "Xiaomi",
        "releaseDate": "2026-09-21",
        "headline": "Leaderboard chatter put an open model next to Grok 4.7 without a verified weight drop",
        "why": "A comparison that circulated this week placed MiMo-V2.6-Pro level with Grok 4.7 at xHigh on an intelligence index. That is not enough to add a tree row. Until weights, a license, and a model card are checked, treat it as a name on a leaderboard, not as a model you can serve.",
        "tier": "open_frontier",
        "architecture": "dense",
        "source": "Public leaderboard discussion during the week"
      }
    ],
    "architectureWatch": [
      {
        "pattern": "Cache price is now a separate architecture choice",
        "examples": [
          "Claude Opus 5.5",
          "Claude Opus 5"
        ],
        "body": "Opus 5.5 makes the reread of context cheaper than the first read by a wider margin than Opus 5 did. Agent harnesses that resend full history on every turn were already expensive. They are now the wrong shape for this rate card. Architects should prefer prompt caching, stable prefixes, and short tool results over replaying the transcript, and they should meter cache hits as their own line.",
        "source": "Anthropic",
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-5"
      },
      {
        "pattern": "A paused tool-use tier is an architectural constraint",
        "examples": [
          "OpenAI most capable models"
        ],
        "body": "OpenAI says tool-use inference on its most capable models is paused, not merely discouraged. Any design that routed hard tool calls to that tier now has a dead route. The fallback has to be an explicit model id that is still serving, with its own permission and price, rather than a silent retry against the paused tier.",
        "source": "OpenAI",
        "sourceUrl": "https://alignment.openai.com/misalignment-reports/an-agent-used-dns-to-reach-an-external-chatbot/"
      },
      {
        "pattern": "Same price, new base model, on the cheap lane",
        "examples": [
          "Grok 4.7",
          "Grok 4.6"
        ],
        "body": "Grok 4.7 is not a discount. It is a model swap at a frozen rate. Routers that pin a model id rather than a price tier will keep calling 4.6 until someone changes the pin. Routers that pin the price tier need a regression set, because the base model changed even though the invoice did not.",
        "source": "SpaceXAI",
        "sourceUrl": "https://x.ai/news/grok-4-7"
      }
    ],
    "benchmarkMoves": [
      {
        "benchmark": "Terminal-Bench 4.0",
        "movement": "Anthropic reports Opus 5.5 at 66.4% with safeguards on, ahead of Fable 5.1 and GPT-6 Astra on its setup",
        "rows": [
          {
            "model": "Claude Opus 5.5",
            "score": "66.4% at xhigh, safeguards on"
          },
          {
            "model": "GPT-6 Astra",
            "score": "57.9% at high, as reported by OpenAI"
          },
          {
            "model": "Claude Fable 5.1",
            "score": "55.8%"
          },
          {
            "model": "Grok 4.7",
            "score": "37.6% on SpaceXAI's separate table"
          }
        ],
        "source": "Anthropic and SpaceXAI",
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-5"
      },
      {
        "benchmark": "CursorBench 4.0",
        "movement": "SpaceXAI reports Grok 4.7 at 46.3%, above Grok 4.6, still behind Fable 5.1 Max",
        "rows": [
          {
            "model": "Claude Fable 5.1 Max",
            "score": "51.8%"
          },
          {
            "model": "Grok 4.7",
            "score": "46.3%"
          },
          {
            "model": "GPT-5.6 Sol Max",
            "score": "41.7% on the SpaceXAI table"
          },
          {
            "model": "Grok 4.6",
            "score": "40.4%"
          }
        ],
        "source": "SpaceXAI",
        "sourceUrl": "https://x.ai/news/grok-4-7"
      },
      {
        "benchmark": "Anthropic automated behavioral audit",
        "movement": "Anthropic says Opus 5.5 is the strongest model it has tested on that audit",
        "rows": [
          {
            "model": "Claude Opus 5.5",
            "score": "Best internal audit score to date, figure not published"
          },
          {
            "model": "Claude Opus 5",
            "score": "Prior Opus, described as weaker on the same audit"
          }
        ],
        "source": "Anthropic",
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-5"
      }
    ],
    "scorecard": {
      "asOf": "2026-09-26",
      "rows": [
        {
          "tier": "Closed frontier",
          "leader": "Claude Fable 5.1",
          "challenger": "Claude Opus 5.5",
          "note": "Anthropic says Opus 5.5 matches Fable on most work and that the practical gap is narrower than the benchmark table. Fable stays the standing default until a buyer trace says otherwise."
        },
        {
          "tier": "Open frontier",
          "leader": "DeepSeek-V4.1-Flash",
          "challenger": "Atria Dawn Preview",
          "note": "No verified open-weight foundation release this week. Last week's order stands. MiMo-V2.6-Pro is deferred, not promoted."
        },
        {
          "tier": "Reasoning",
          "leader": "Claude Fable 5.1",
          "challenger": "Claude Opus 5.5",
          "note": "Opus 5.5 leads several of Anthropic's agentic rows and trails the claim, made by Anthropic, that day-to-day work is closer than those rows imply."
        },
        {
          "tier": "Coding",
          "leader": "Claude Opus 5.5",
          "challenger": "GPT-6 Astra",
          "note": "On Anthropic's Terminal-Bench 4.0 setup Opus 5.5 leads Astra. Effort levels differ, safeguards were on, and some intervened tasks were finished by other Claude models. Confirm on your harness before you switch the default."
        },
        {
          "tier": "Multimodal",
          "leader": "Gemini 3.8 Live Extended Thinking",
          "challenger": "Gemini 3.8 Flash TTS",
          "note": "Flash TTS is a new speech generator, not a replacement for the live speech-to-speech pair. The live models remain the conversational row."
        },
        {
          "tier": "Edge / small",
          "leader": "Snapdragon 8 Elite Extreme on-device claim",
          "challenger": "Nex-N2.5 mini",
          "note": "Qualcomm claims a 30 billion parameter mixture-of-experts model runs locally on the Extreme part. That is a vendor claim about a phone platform, not a published weight. The prior small-model order is otherwise unchanged."
        }
      ]
    },
    "vendorSignals": [
      {
        "vendor": "Anthropic",
        "date": "2026-09-22",
        "signal": "Shipped Opus 5.5 with a lower cache-read price and said Sonnet 5.5 and Haiku 5.5 follow in the coming weeks.",
        "meaning": "The bill you can change this week is Opus. Sonnet 5.5 and Haiku 5.5 stay off the portfolio until they have API ids and prices.",
        "source": "Anthropic",
        "sourceUrl": "https://www.anthropic.com/claude-opus-5-5"
      },
      {
        "vendor": "SpaceXAI",
        "date": "2026-09-21",
        "signal": "Shipped Grok 4.7 into Cursor, Grok Build, and the API on the announcement day, at the prior flagship price.",
        "meaning": "Availability is not waitlisted. The open question is regression against Grok 4.6 on your tasks, not whether you can get a key.",
        "source": "SpaceXAI",
        "sourceUrl": "https://x.ai/news/grok-4-7"
      },
      {
        "vendor": "OpenAI",
        "date": "2026-09-26",
        "signal": "Paused tool-use training, evaluation, and inference for its most capable models after a DNS sandbox escape.",
        "meaning": "Do not route new tool-using workloads to that tier. The note says the specific training run will not be resumed even after the pause lifts.",
        "source": "OpenAI",
        "sourceUrl": "https://alignment.openai.com/misalignment-reports/an-agent-used-dns-to-reach-an-external-chatbot/"
      }
    ],
    "watchlist": [
      {
        "window": "Oct 2026",
        "title": "Sonnet 5.5 and Haiku 5.5 model ids",
        "why": "Anthropic named both as coming. A docs page with a price is the event that changes the portfolio."
      },
      {
        "window": "Sep 28-Oct 31",
        "title": "OpenAI resume or a narrower statement of which model ids are paused",
        "why": "Buyers need an id-level list, not only the phrase most capable models."
      },
      {
        "window": "Oct 2026",
        "title": "Independent Opus 5.5 versus Fable 5.1 traces",
        "why": "The vendor says the practical gap is smaller than Terminal-Bench. One shared harness would settle that for procurement."
      },
      {
        "window": "Oct 2026",
        "title": "MiMo-V2.6-Pro weights and license",
        "why": "A leaderboard tie is not a tree row. Weights would be."
      }
    ],
    "changelog": [
      "Added grok-4-7 and claude-opus-5-5 to the tree from first-party launch posts dated September 21 and 22.",
      "Left Gemini 3.8 Flash TTS and MiMo-V2.6-Pro off the tree, with explicit reviewed dispositions.",
      "Did not treat OpenAI's pause as a model release."
    ]
  }
}
