{
  "_meta": {
    "publication": "The Model Pulse",
    "schemaVersion": "2026.05.02",
    "generatedAt": "2026-09-08T22:45:10.946Z",
    "canonicalUrl": "https://brianletort.ai/industry/models/2026-W25",
    "markdownUrl": "https://brianletort.ai/industry/models/2026-W25/llm.md",
    "pdfUrl": "https://brianletort.ai/downloads/model-pulse-2026-W25.pdf",
    "treeUrl": "https://brianletort.ai/industry/tree",
    "sourceFile": "src/data/industry/models/2026-W25.ts"
  },
  "issue": {
    "slug": "2026-W25",
    "isoYear": 2026,
    "isoWeek": 25,
    "issueNumber": 9,
    "publishedAt": "2026-06-20",
    "cadence": "weekly",
    "periodLabel": "Week 25 of 2026",
    "bigRead": {
      "headline": "Open weights took the lead while the closed frontier stalled — and the benchmark itself was rewritten.",
      "body": "W25 was a loud week for open weights and a quiet one at the closed frontier. Z.ai shipped GLM-5.2 under a genuine MIT license — a ~744B / ~40B-active sparse-attention MoE with 1M context that independent testing (VentureBeat) says beats GPT-5.5 on several long-horizon coding benchmarks at roughly one-sixth the cost, and that Artificial Analysis now cites as the leading open-weight model. MiniMax-M3's sparse-attention weights matured in-window with an arXiv report validating its efficiency claims, though its non-OSI Community License gates commercial use. The closed frontier, by contrast, marked time: no GA from OpenAI (GPT-5.6 remains rumor) or xAI, Gemini 3.5 Pro slipped from June to July, and Anthropic's Claude Fable 5 stayed government-suspended the entire week (Opus 4.8 is the working leader). The third shift was measurement itself — Artificial Analysis rebased its Intelligence Index to v4.1, re-weighting the industry's headline benchmark around agentic tasks, so scores are no longer back-comparable to v4.0. The procurement implication: open self-host is now a live coding option, not a hedge; teams should pilot MIT-licensed GLM-5.2, read every 'open' license carefully (GLM-5.2 MIT vs MiniMax Community), and re-baseline evaluations on the agentic v4.1 index while keeping closed-frontier fallbacks given the demonstrated availability risk."
    },
    "treeDelta": {
      "summary": "Two W25 additions: GLM-5.2 (MIT open-weight frontier-adjacent MoE that took the open lead) and MiniMax-M3 (sparse-attention multimodal MoE, weights matured with arXiv verification).",
      "added": [
        "glm-5-2",
        "minimax-m3"
      ],
      "updated": [],
      "note": "No new closed frontier model entered the tree: GPT-5.6 is rumor only, Gemini 3.5 Pro slipped to July, and Claude Fable 5 (added W24) stayed suspended. ByteDance's seed-2.1-pro-preview is excluded as an undisclosed preview."
    },
    "frontierMovements": [
      {
        "name": "Claude Fable 5 (still suspended)",
        "vendor": "Anthropic",
        "releaseDate": "2026-06-09",
        "headline": "Remained government-suspended for the full week; #1 on the rebased Artificial Analysis Index (60) but unavailable, so Opus 4.8 (56) is the top available closed model",
        "why": "The closed-frontier leader on paper is unusable in practice for a second week, which keeps the availability/sovereign risk live. Architects should treat the AA top score as aspirational and standardize on the top available model (Opus 4.8) with fallback routing, not on a suspended SKU.",
        "tier": "frontier",
        "architecture": "reasoning",
        "source": "Anthropic; Artificial Analysis Intelligence Index v4.1",
        "sourceUrl": "https://www.anthropic.com/news/claude-fable-5-mythos-5"
      },
      {
        "name": "Gemini 3.5 Pro (slipped to July)",
        "vendor": "Google DeepMind",
        "releaseDate": "2026-07 target",
        "headline": "GA slipped from June to July: still a limited Vertex preview with no public model card, pricing, or independent benchmark vs Fable 5/Opus 4.8",
        "why": "A frontier movement by absence for the third consecutive issue. Buyers should keep an evaluation slot ready but not pause current baselines; the closed frontier's cadence is visibly slipping while open weights accelerate.",
        "tier": "frontier",
        "architecture": "reasoning",
        "source": "Business Insider; Google",
        "sourceUrl": "https://www.businessinsider.com/google-3-5-pro-july-release-tokens-ai-agents-model-2026-6"
      }
    ],
    "openWeights": [
      {
        "modelId": "glm-5-2",
        "name": "GLM-5.2",
        "vendor": "Z.ai",
        "releaseDate": "2026-06-16",
        "headline": "MIT-licensed ~744B / ~40B-active sparse-attention MoE with 1M context; independent testing says it beats GPT-5.5 on long-horizon coding at ~1/6 the cost and Artificial Analysis cites it as the top open-weight model",
        "why": "GLM-5.2 is the week's most consequential release: a truly permissive (MIT, no regional limits) frontier-adjacent model that is self-hostable and sovereignty-friendly for long-context agentic coding. Architects should pilot it for self-host/private coding workloads and use its ~$1.40/$4.40 per-MTok API as a pricing benchmark against closed flagships.",
        "tier": "open_frontier",
        "architecture": "moe",
        "source": "Z.ai blog; VentureBeat; Hugging Face",
        "sourceUrl": "https://z.ai/blog/glm-5.2"
      },
      {
        "modelId": "minimax-m3",
        "name": "MiniMax-M3",
        "vendor": "MiniMax",
        "releaseDate": "2026-06-12",
        "headline": "428B / ~23B-active sparse-attention MoE with native multimodality and 1M context; weights matured with an arXiv report verifying ~9x/15x prefill/decode efficiency, under a non-OSI Community License",
        "why": "MiniMax-M3 combines frontier-adjacent coding, genuine 1M context, and native multimodality in one downloadable checkpoint — but the MiniMax Community License gates commercial use, so 'open weights' here does not mean free to deploy. Teams should verify the efficiency claims via the arXiv report and clear the license before planning a commercial deployment.",
        "tier": "open_frontier",
        "architecture": "moe",
        "source": "TechTimes; Hugging Face; arXiv:2606.13392",
        "sourceUrl": "https://www.techtimes.com/articles/318622/20260618/minimax-m3-takes-open-weight-ai-lead-sparse-attention-architecture-now-verified.htm"
      }
    ],
    "architectureWatch": [
      {
        "pattern": "Open weights close the cost gap",
        "examples": [
          "GLM-5.2 (MIT)",
          "MiniMax-M3",
          "Grok 4.3 on Bedrock"
        ],
        "body": "Frontier-adjacent capability is collapsing toward commodity inference pricing. GLM-5.2's MIT weights reportedly match or beat GPT-5.5 on long-horizon coding at ~1/6 the cost, and API list prices (GLM-5.2 ~$1.40/$4.40 per MTok; Grok 4.3 $1.25/$2.50 on Bedrock) keep falling. Procurement should pilot open self-host for routine and long-context coding and reserve closed flagships for the highest-risk reasoning.",
        "source": "Z.ai; VentureBeat; Amazon Bedrock"
      },
      {
        "pattern": "The headline benchmark pivots to agents",
        "examples": [
          "Artificial Analysis Intelligence Index v4.1",
          "LMArena Agent Arena"
        ],
        "body": "Artificial Analysis rebased its Intelligence Index to v4.1, re-weighting around agentic tasks (GDPval-AA v2 at 20%, Terminal-Bench, banking agents) and dropping a saturated benchmark, while LMArena's Agent Arena scores behavioral signals (retries, steerability) rather than preference votes. Boards comparing models on 'the AA Index' must note v4.1 scores are not back-comparable to v4.0; re-baseline evaluation harnesses now.",
        "source": "Artificial Analysis; arena.ai changelog"
      },
      {
        "pattern": "License divergence within 'open'",
        "examples": [
          "GLM-5.2 (MIT)",
          "MiniMax-M3 (Community License)"
        ],
        "body": "Two of the week's open releases sit on opposite ends of the permissiveness spectrum: GLM-5.2 under MIT with no regional limits versus MiniMax-M3 under a Community License that gates commercial use. For enterprise adoption, 'open weights' does not equal 'free to deploy commercially' — legal and procurement should read the actual license before standardizing on a model.",
        "source": "Z.ai; MiniMax Hugging Face card"
      }
    ],
    "benchmarkMoves": [
      {
        "benchmark": "Artificial Analysis Intelligence Index v4.1",
        "movement": "Methodology rebased around agentic tasks (Jun 15); leaders Fable 5 = 60 (top but suspended), Opus 4.8 = 56 (top available), GPT-5.5 = 55; scores not back-comparable to v4.0",
        "rows": [
          {
            "model": "Claude Fable 5 (suspended)",
            "score": "60"
          },
          {
            "model": "Claude Opus 4.8 (top available)",
            "score": "56"
          },
          {
            "model": "GPT-5.5",
            "score": "55"
          }
        ],
        "source": "Artificial Analysis",
        "sourceUrl": "https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-1"
      },
      {
        "benchmark": "Open-weight leaderboard",
        "movement": "GLM-5.2 took the open-weight lead in-window; MiniMax-M3 and DeepSeek V4 Pro sit at ~44 on the rebased index",
        "rows": [
          {
            "model": "GLM-5.2",
            "score": "top open (AA v4.1)"
          },
          {
            "model": "MiniMax-M3",
            "score": "44"
          },
          {
            "model": "DeepSeek V4 Pro",
            "score": "44"
          }
        ],
        "source": "Artificial Analysis; VentureBeat",
        "sourceUrl": "https://venturebeat.com/technology/z-ais-open-weights-glm-5-2-beats-gpt-5-5-on-multiple-long-horizon-coding-benchmarks-for-1-6th-the-cost"
      }
    ],
    "scorecard": {
      "asOf": "2026-06-20",
      "rows": [
        {
          "tier": "Closed frontier",
          "leader": "Claude Opus 4.8",
          "challenger": "GPT-5.5",
          "note": "Fable 5 leads the AA v4.1 index (60) but stayed suspended all week; Opus 4.8 (56) is the top available closed model."
        },
        {
          "tier": "Open frontier",
          "leader": "GLM-5.2",
          "challenger": "MiniMax-M3",
          "note": "GLM-5.2's MIT release took the open-weight lead in-window; MiniMax-M3 contends but is gated by a non-OSI license."
        },
        {
          "tier": "Reasoning",
          "leader": "Claude Opus 4.8",
          "challenger": "GPT-5.5",
          "note": "Closed reasoning leadership steady among available models while Gemini 3.5 Pro slipped to July."
        },
        {
          "tier": "Coding",
          "leader": "Claude Opus 4.8",
          "challenger": "GLM-5.2",
          "note": "Open weights are closing fast on long-horizon coding; GLM-5.2 reportedly beats GPT-5.5 at ~1/6 the cost."
        },
        {
          "tier": "Multimodal",
          "leader": "Gemini 3.1 Pro",
          "challenger": "MiniMax-M3",
          "note": "Gemini remains the general multimodal reference; MiniMax-M3 adds native-multimodal open weights."
        },
        {
          "tier": "Edge / small",
          "leader": "Mellum2",
          "challenger": "North Mini Code",
          "note": "Efficient open coding/sub-agent models unchanged in-window; the week's open action was at the frontier-adjacent tier."
        }
      ]
    },
    "vendorSignals": [
      {
        "vendor": "Z.ai",
        "date": "2026-06-16",
        "signal": "Released GLM-5.2 under MIT with API pricing ~$1.40/$4.40 per MTok (~1/6 of comparable frontier)",
        "meaning": "A tier-1 permissive open release at commodity pricing pressures every closed flagship's price/value story. Procurement should use GLM-5.2 as a negotiating anchor and pilot it for self-host coding; investors should treat open-weight pricing as a structural deflationary force on inference.",
        "source": "Z.ai; DataNorth",
        "sourceUrl": "https://z.ai/blog/glm-5.2"
      },
      {
        "vendor": "Artificial Analysis",
        "date": "2026-06-15",
        "signal": "Rebased the Intelligence Index to v4.1, re-weighting around agentic workloads; v4.1 scores are not back-comparable to v4.0",
        "meaning": "The industry's headline benchmark now measures agentic capability, not static Q&A. Boards and architects must re-baseline model comparisons on v4.1 and avoid mixing old and new index numbers in procurement decisions.",
        "source": "Artificial Analysis",
        "sourceUrl": "https://artificialanalysis.ai/articles/artificial-analysis-intelligence-index-v4-1"
      },
      {
        "vendor": "xAI",
        "date": "2026-06-15",
        "signal": "Grok 4.3 went GA on Amazon Bedrock ($1.25/$2.50 per MTok, 1M context), making xAI the third independent frontier lab on Bedrock alongside Anthropic and OpenAI",
        "meaning": "CIOs can now evaluate all three independent US frontier labs under one IAM and billing surface. The caveat is a non-standard endpoint and a context-window pricing cliff above 200K tokens; this is distribution, not a new capability tier.",
        "source": "DigitalApplied; Memeburn",
        "sourceUrl": "https://www.digitalapplied.com/blog/grok-4-3-amazon-bedrock-enterprise-launch-2026-guide"
      },
      {
        "vendor": "Anthropic",
        "date": "2026-06-15",
        "signal": "Claude Fable 5 and Mythos 5 remained government-suspended all week; the planned Jun 23 usage-credit subscription change is moot while access is off",
        "meaning": "The top-tier closed model's availability is still a sovereign/regulatory variable, not an SLA. Buyers should keep Opus 4.8/Sonnet fallbacks wired and avoid single-sourcing the frontier for production-critical paths.",
        "source": "Anthropic",
        "sourceUrl": "https://www.anthropic.com/news/claude-fable-5-mythos-5"
      }
    ],
    "watchlist": [
      {
        "window": "July",
        "title": "Gemini 3.5 Pro GA",
        "why": "Pro slipped to July. Its GA and first independent AA v4.1 pass will show whether Google can re-take a frontier lead now contested by both Opus 4.8 and a surging open-weight field."
      },
      {
        "window": "Jun-Aug",
        "title": "Claude Fable 5 / Mythos 5 restoration",
        "why": "Restoration terms (geo-gating, KYC, or a permanent civilian/government capability split) will set the precedent for sovereign access risk and decide whether Fable 5 re-enters the available scorecard."
      },
      {
        "window": "Jun-Aug",
        "title": "GLM-5.2 adoption and independent SWE-Bench replication",
        "why": "Downloads, integrations, and third-party benchmark replication will show whether MIT-licensed open weights become production substrate and force closed-flagship price cuts."
      },
      {
        "window": "Jun-Jul",
        "title": "GPT-5.6 / next OpenAI flagship",
        "why": "Codenames and prediction markets pointed to a launch just after this window. A real system card would re-set the closed frontier and test whether OpenAI answers the open-weight cost pressure."
      }
    ],
    "changelog": [
      "Added GLM-5.2 and MiniMax-M3 to the LLM tree; reframed W25 around open weights taking the lead while the closed frontier stalled (Fable 5 suspended, Gemini slipped) and Artificial Analysis rebased to the agentic v4.1 index."
    ]
  }
}
