{
  "_meta": {
    "publication": "The Model Pulse",
    "schemaVersion": "2026.05.02",
    "generatedAt": "2026-09-08T22:45:10.738Z",
    "canonicalUrl": "https://brianletort.ai/industry/models/2026-04-recap",
    "markdownUrl": "https://brianletort.ai/industry/models/2026-04-recap/llm.md",
    "pdfUrl": "https://brianletort.ai/downloads/model-pulse-2026-04-recap.pdf",
    "treeUrl": "https://brianletort.ai/industry/tree",
    "sourceFile": "src/data/industry/models/2026-04-recap.ts"
  },
  "issue": {
    "slug": "2026-04-recap",
    "isoYear": 2026,
    "isoWeek": 17,
    "issueNumber": 1,
    "publishedAt": "2026-04-25",
    "cadence": "monthly_recap",
    "periodLabel": "April 2026 Recap",
    "bigRead": {
      "headline": "April rewrote the floor: open weights crossed the closed frontier, and the safety-gated frontier became a procurement criterion.",
      "body": "April 2026 was the most productive month in frontier model history. Eleven model rows landed in the LLM Evolutionary Tree across eight vendors, including the first open-weights MoE to score frontier-class on SWE-Bench Pro coding (Kimi K2.6 at 58.6 and GLM-5.1 at 58.4, both above GPT-5.4 at 57.7 and Claude Opus 4.6 at 57.3). DeepSeek V4 shipped in two configurations under MIT license with 1M-token context and 1.6T total parameters, moving on-prem coding from 'can we?' to 'which workload first?' Anthropic withheld its Mythos flagship on cyber-capability grounds after UK AISI confirmed autonomous offensive capability — an inflection that turns capability gating from a research-org concern into a procurement diligence requirement. The closed frontier responded: GPT-5.5 (Apr 23) and Claude Opus 4.7 (Apr 16) reset the closed-source ceiling, with adaptive-thinking and ultra-long-context as the new defaults. Read together, April was the month the canopy widened on three axes at once: open-vs-closed parity on coding, reasoning-as-default at every tier, and capability gating as a market signal."
    },
    "treeDelta": {
      "summary": "11 model rows added to the tree in April, spanning frontier closed releases, open-frontier MoEs, gated safety-class previews, and reasoning-tier successors. The decoder-only zone widened most; the multimodal and reasoning branches absorbed every major release.",
      "added": [
        "claude-opus-4-6",
        "qwen-3-6-plus",
        "claude-mythos",
        "glm-5-1",
        "claude-opus-4-7",
        "qwen-3-6-35b-a3b",
        "kimi-k2-6",
        "gpt-5-5",
        "deepseek-v4-pro",
        "deepseek-v4-flash",
        "deepseek-r2"
      ],
      "updated": [
        "claude-opus-4-5",
        "kimi-k2-5",
        "gpt-5-4",
        "qwen-3-5"
      ],
      "note": "The Anthropic line gained two flagship releases plus one gated preview in a single month. DeepSeek shipped V4 in two scales (Pro and Flash) plus a same-month R2 reasoning successor — a cadence no other vendor matched in April."
    },
    "frontierMovements": [
      {
        "modelId": "claude-opus-4-7",
        "name": "Claude Opus 4.7",
        "vendor": "Anthropic",
        "releaseDate": "2026-04-16",
        "headline": "Adaptive-thinking flagship with ultra-long context and full agentic stack (computer-use + MCP + tool use).",
        "why": "Reset the closed-frontier reasoning ceiling. The agentic surface area (computer-use, MCP, tool-use) makes Opus 4.7 the default reference for closed-frontier procurement when adaptive thinking and long-horizon agent runs are required. Pair with Sonnet 4.6 for the cost-tiered deployment.",
        "tier": "frontier",
        "architecture": "agentic",
        "source": "anthropic.com release notes, model card"
      },
      {
        "modelId": "gpt-5-5",
        "name": "GPT-5.5",
        "vendor": "OpenAI",
        "releaseDate": "2026-04-23",
        "headline": "Frontier reasoning flagship with multimodal and ultra-long-context as defaults.",
        "why": "OpenAI's response to the V4 / Opus 4.7 surge. Sets the closed-frontier reasoning bar that the open-frontier challengers (V4 Pro, K2.6, GLM-5.1) measure against on SWE-Bench Pro and GPQA. The 5.x family now spans Instant, Thinking, Pro, mini, and nano tiers — useful for matching workload class to model class.",
        "tier": "frontier",
        "architecture": "reasoning",
        "source": "openai.com blog, GPT-5.5 system card"
      },
      {
        "modelId": "claude-mythos",
        "name": "Claude Mythos Preview",
        "vendor": "Anthropic",
        "releaseDate": "2026-04-07",
        "headline": "Frontier-class capability withheld from public access on cyber-capability grounds.",
        "why": "The first major model gated by its own developer for safety reasons after independent (UK AISI) red-teaming confirmed autonomous offensive capability. Vendor-risk frameworks now need a capability-gate criterion alongside availability SLAs — diligence on 'what could the next release do that the current one cannot?' is no longer optional.",
        "tier": "specialist",
        "architecture": "agentic",
        "source": "anthropic.com, red.anthropic.com, aisi.gov.uk"
      },
      {
        "modelId": "claude-opus-4-6",
        "name": "Claude Opus 4.6",
        "vendor": "Anthropic",
        "releaseDate": "2026-04-02",
        "headline": "Predecessor to 4.7; held the SWE-Bench Pro closed-frontier line at 57.3 mid-month.",
        "why": "Useful as the April 02 baseline against which open-weights eventually moved past mid-month. Procurement teams running 4.5 should plan upgrade to 4.7 directly; 4.6 will likely move to maintenance status before mid-May.",
        "tier": "frontier",
        "architecture": "reasoning",
        "source": "anthropic.com release notes"
      }
    ],
    "openWeights": [
      {
        "modelId": "deepseek-v4-pro",
        "name": "DeepSeek-V4 Pro",
        "vendor": "DeepSeek AI",
        "releaseDate": "2026-04-24",
        "headline": "1.6T MoE, 1M context, MIT license, frontier-class on coding.",
        "why": "The headline release of April. First open-weights model with frontier-class SWE-Bench Pro performance under a permissive license. Procurement teams blocked on closed-source data residency or licensing constraints now have an open frontier alternative for coding workloads. The on-prem floor moved up; the merchant-vs-self-host fork is now real.",
        "tier": "open_frontier",
        "architecture": "moe",
        "source": "huggingface.co/deepseek, deepseek.com"
      },
      {
        "modelId": "deepseek-v4-flash",
        "name": "DeepSeek-V4 Flash",
        "vendor": "DeepSeek AI",
        "releaseDate": "2026-04-24",
        "headline": "Cost-tiered V4 sibling for high-throughput inference on smaller fabric.",
        "why": "Open-weights answer to GPT-4o-mini and Claude Sonnet at the inference-cost tier. Useful for batch coding agents and high-volume serving where Pro is overkill. Same MIT license, same 1M context, lower active-parameter cost per token.",
        "tier": "open_frontier",
        "architecture": "moe",
        "source": "huggingface.co/deepseek model card"
      },
      {
        "modelId": "kimi-k2-6",
        "name": "Kimi K2.6",
        "vendor": "Moonshot AI",
        "releaseDate": "2026-04-20",
        "headline": "58.6 on SWE-Bench Pro — the first open-weights model to top the closed leaders on coding.",
        "why": "The benchmark moment. K2.6's coding score sits above GPT-5.4 (57.7) and Claude Opus 4.6 (57.3) under an open-weights license. For coding-heavy procurement reads, K2.6 is the new default open-frontier baseline; closed-source premium needs new justification beyond raw score.",
        "tier": "open_frontier",
        "architecture": "moe",
        "source": "moonshot.cn, Artificial Analysis SWE-Bench Pro"
      },
      {
        "modelId": "glm-5-1",
        "name": "GLM-5.1",
        "vendor": "Z.AI (Zhipu)",
        "releaseDate": "2026-04-08",
        "headline": "58.4 on SWE-Bench Pro with multi-agent-native architecture and ultra-long context.",
        "why": "Second open-weights model to clear the closed-frontier coding bar this month. The multi-agent-native design point makes GLM-5.1 the open-source reference for agent-of-agents deployments — relevant when coordination overhead matters more than single-shot completion.",
        "tier": "open_frontier",
        "architecture": "agentic",
        "source": "z.ai, GLM-5.1 model card"
      },
      {
        "modelId": "qwen-3-6-35b-a3b",
        "name": "Qwen3.6-35B-A3B",
        "vendor": "Alibaba",
        "releaseDate": "2026-04-16",
        "headline": "Open-weights MoE successor to Qwen3.5; agentic + multimodal at deployable scale.",
        "why": "The most deployable of April's open releases on commodity inference fabric. 35B total / 3B active makes Qwen3.6-A3B the open multimodal pick when V4 Pro is too large and K2.6 is single-modality. Pairs naturally with Qwen3.6-Plus on the closed-source side for hybrid deployments.",
        "tier": "open_frontier",
        "architecture": "moe",
        "source": "qwenlm.github.io, model card"
      },
      {
        "modelId": "deepseek-r2",
        "name": "DeepSeek-R2",
        "vendor": "DeepSeek AI",
        "releaseDate": "2026-04",
        "headline": "Open-weights reasoning successor to R1, frontier-class long-context.",
        "why": "Same vendor shipping V4 (general) and R2 (reasoning) inside one calendar month is the cadence story. R2 closes the open-vs-closed reasoning gap that R1 opened in 2025 — open-weights reasoning is now a separate procurement category, not a niche.",
        "tier": "reasoning",
        "architecture": "reasoning",
        "source": "deepseek.com, huggingface.co/deepseek"
      }
    ],
    "architectureWatch": [
      {
        "pattern": "Reasoning becomes the default mode, not a separate model.",
        "examples": [
          "Claude Opus 4.7 (adaptive thinking)",
          "GPT-5.5",
          "DeepSeek-R2",
          "GLM-5.1"
        ],
        "body": "April's flagships ship reasoning behavior as a default rather than a separate o-series-style fork. Adaptive thinking, where the model decides how much to deliberate, is now the closed-frontier expectation; explicit reasoning toggles look dated by month-end. Procurement consequence: the reasoning-vs-non-reasoning split is collapsing — assume reasoning is on, budget tokens accordingly.",
        "source": "Vendor model cards (Anthropic, OpenAI, DeepSeek, Z.AI)"
      },
      {
        "pattern": "Million-token context as baseline.",
        "examples": [
          "DeepSeek-V4 Pro (1M)",
          "Kimi K2.6 (1M+)",
          "Claude Opus 4.7 (ultra-long)",
          "GPT-5.5 (ultra-long)"
        ],
        "body": "Every flagship released in April clears the 1M-token threshold. Long-context is no longer a differentiator at the frontier; it is the floor. Workloads that previously required RAG architectures may now be expressible as direct in-context loads — re-evaluate the retrieval layer when re-baselining for V4 / K2.6 / Opus 4.7.",
        "source": "Model cards; HuggingFace technical reports"
      },
      {
        "pattern": "Agentic surface area as a first-class capability.",
        "examples": [
          "Claude Opus 4.7 (computer-use, MCP, tool-use)",
          "GLM-5.1 (multi-agent-native)",
          "Qwen3.6-Plus (agentic)",
          "Kimi K2.6 (agentic-ready)"
        ],
        "body": "Agentic isn't a benchmark category anymore; it is a model property declared on the card. Computer-use, MCP support, and multi-agent coordination ship with the flagship rather than as a downstream wrapper. Architectural read: the boundary between 'model' and 'agent runtime' is dissolving — procurement should evaluate the full agent surface, not just the LM weights.",
        "source": "Vendor model cards; Model Context Protocol announcements"
      },
      {
        "pattern": "Capability gating as a market signal.",
        "examples": [
          "Claude Mythos Preview (withheld)"
        ],
        "body": "Anthropic withholding Mythos after UK AISI red-team findings is the first time a major lab gated a frontier-class release on its own initiative for cyber-capability reasons. The signal: capability evaluation now happens before public release, and 'gated' is a status that procurement teams need to track in vendor risk frameworks alongside 'available' and 'deprecated.'",
        "source": "anthropic.com, red.anthropic.com, aisi.gov.uk"
      }
    ],
    "benchmarkMoves": [
      {
        "benchmark": "SWE-Bench Pro (coding)",
        "movement": "Open weights overtook closed for the first time. Two open-weights models clear both closed leaders.",
        "rows": [
          {
            "model": "Kimi K2.6",
            "score": "58.6"
          },
          {
            "model": "GLM-5.1",
            "score": "58.4"
          },
          {
            "model": "GPT-5.4",
            "score": "57.7"
          },
          {
            "model": "Claude Opus 4.6",
            "score": "57.3"
          }
        ],
        "source": "Artificial Analysis SWE-Bench Pro leaderboard, vendor model cards"
      },
      {
        "benchmark": "Long-context retrieval (1M+ tokens, vendor-reported)",
        "movement": "Million-token context lands at the frontier with sub-2% retrieval-error rates across multiple vendors.",
        "rows": [
          {
            "model": "DeepSeek-V4 Pro",
            "score": "1M ctx, 1.4% err"
          },
          {
            "model": "Kimi K2.6",
            "score": "1M ctx, 1.7% err"
          },
          {
            "model": "Claude Opus 4.7",
            "score": "ultra-long, vendor-reported"
          },
          {
            "model": "GPT-5.5",
            "score": "ultra-long, vendor-reported"
          }
        ],
        "source": "Vendor technical reports; HuggingFace evaluations"
      },
      {
        "benchmark": "Agentic tool-use (vendor + third-party)",
        "movement": "Closed flagships still lead agentic tool-use; open frontier is one tier behind but closing.",
        "rows": [
          {
            "model": "Claude Opus 4.7",
            "score": "Closed leader"
          },
          {
            "model": "GPT-5.5",
            "score": "Closed challenger"
          },
          {
            "model": "GLM-5.1",
            "score": "Open leader (multi-agent-native)"
          },
          {
            "model": "DeepSeek-V4 Pro",
            "score": "Open challenger"
          }
        ],
        "source": "Vendor agent benchmarks; community evaluations"
      }
    ],
    "scorecard": {
      "asOf": "2026-04-25",
      "rows": [
        {
          "tier": "Closed frontier",
          "leader": "Claude Opus 4.7",
          "challenger": "GPT-5.5",
          "note": "Adaptive thinking is the closed-source differentiator this month."
        },
        {
          "tier": "Open frontier",
          "leader": "DeepSeek-V4 Pro",
          "challenger": "Kimi K2.6",
          "note": "MIT license + 1M context + frontier coding score is the new open baseline."
        },
        {
          "tier": "Reasoning",
          "leader": "DeepSeek-R2",
          "challenger": "GPT-5.5 (thinking mode)",
          "note": "Open-weights reasoning is now its own procurement category."
        },
        {
          "tier": "Coding",
          "leader": "Kimi K2.6",
          "challenger": "GLM-5.1",
          "note": "Open-weights tops both closed leaders on SWE-Bench Pro this month."
        },
        {
          "tier": "Multimodal",
          "leader": "Gemini 3.1 Pro",
          "challenger": "Qwen3.6-Plus",
          "note": "Closed leader from Q1 still holds; Qwen3.6 narrows the gap on the open side."
        },
        {
          "tier": "Edge / small",
          "leader": "Phi-4-mini",
          "challenger": "Gemma 4",
          "note": "Q1 leader carries through April; no major edge-tier release this month."
        }
      ]
    },
    "vendorSignals": [
      {
        "vendor": "Anthropic",
        "date": "2026-04-21",
        "signal": "Mythos withheld from public access pending capability-gate review.",
        "meaning": "Sets a precedent for vendor-initiated gating on cyber-capability grounds. Procurement frameworks now need a 'gated' status alongside 'available' / 'deprecated.' Adds a new diligence question: what is the next release the vendor evaluated and chose not to ship?",
        "source": "anthropic.com, red.anthropic.com, aisi.gov.uk"
      },
      {
        "vendor": "DeepSeek AI",
        "date": "2026-04-24",
        "signal": "Two-config V4 launch (Pro + Flash) plus same-month R2 release under MIT.",
        "meaning": "Cadence and licensing both shifted. A monthly cycle with permissive licensing across coding, reasoning, and inference-tier variants is the new open-frontier benchmark for vendor velocity. Closed-source vendors with quarterly cadence have a velocity gap, not just a capability gap.",
        "source": "deepseek.com release notes; huggingface.co model cards"
      },
      {
        "vendor": "OpenAI",
        "date": "2026-04",
        "signal": "GPT-4 family deprecation timeline tightened; 5.x consolidated to five tiers (Instant, Thinking, Pro, mini, nano).",
        "meaning": "Pricing migration for GPT-4-era workloads accelerates through Q2. Customers should plan migration to 5.x within 60 days; the tier consolidation simplifies model selection but compresses the price-vs-capability slope on the low end.",
        "source": "openai.com platform changelog"
      },
      {
        "vendor": "Multiple (open-weights serving)",
        "date": "2026-04",
        "signal": "Inference price cuts of 25-40% across major open-weights serving providers following V4 / K2.6 launches.",
        "meaning": "Open-weights inference economics improved materially this month. Re-baseline cost per token for batch agentic workloads against the new floor; the merchant-vs-self-host break-even moved.",
        "source": "Provider pricing pages; community comparison threads"
      }
    ],
    "watchlist": [
      {
        "window": "May 1-7",
        "title": "Closed-frontier response to V4 / K2.6 coding scores.",
        "why": "Anthropic and OpenAI both have credible reasons to push a coding-tuned point release to recover the SWE-Bench Pro lead. Watch for Opus 4.7-Codex or GPT-5.5-Codex within two weeks."
      },
      {
        "window": "May 1-14",
        "title": "Hyperscaler Q1 prints (MSFT / GOOG / META / AMZN).",
        "why": "AI capex commentary will set the inference-pricing trajectory for the rest of Q2. Tracked in The AI Stack Weekly's capital-flow lens; relevant here as the upstream signal for serving costs."
      },
      {
        "window": "May 5-20",
        "title": "Mythos status update from Anthropic + AISI.",
        "why": "If gating becomes permanent or extended, 'gated' becomes a durable status class. If gating lifts under conditions, a capability-gate playbook emerges — either way, the procurement read changes."
      },
      {
        "window": "May 10-25",
        "title": "Open-source reasoning successor cluster.",
        "why": "DeepSeek-R2 set the bar; Qwen, Z.AI, and Moonshot all have R2-class candidates plausibly within 30 days. Watch for the second open-weights reasoning model to clear o-series parity."
      },
      {
        "window": "May 15-31",
        "title": "First gated-capability standardization proposals.",
        "why": "Mythos creates pressure for a cross-vendor gating taxonomy (capability levels, evaluation protocols, release criteria). Watch for AISI / METR / OpenAI / Anthropic joint statements."
      }
    ],
    "changelog": [
      "Inaugural issue. Cadence: monthly recap covering April 2026 in one read; weekly cadence begins in May.",
      "Tree delta sourced from content/llm-tree/models.yaml April 2026 release rows.",
      "Benchmark scores cite Artificial Analysis leaderboards and vendor model cards as of publish date.",
      "Scorecard reflects April 2026 close; will refresh weekly as new releases land."
    ]
  }
}
