{
  "schema_version": "1.1",
  "id": "s4:https://vercel.com/blog/ai-gateway-production-index-september-2026",
  "slug": "ai-gateway-production-index-september-2026-0drggwl",
  "url": "https://feed7.dev/p/ai-gateway-production-index-september-2026-0drggwl",
  "title": "Open-weight models take 56% of token volume, Astra doubles Fable 5.1 spend",
  "why_included": "Vercel’s gateway data shows open-weight models reached 56% of August token volume while price per token fell, reinforcing workload routing over loyalty to one frontier model.",
  "summary": "Open-weight models rose from **7% of gateway tokens in December to 56% in August**, while accounting for 14% of August spend. Average price per token fell **23.2% in August**, and the median high-volume team paid 7.6% less.",
  "practical_implication": "Treat model choice as routing, not allegiance: run cheaper or open-weight models where evals permit, and reserve frontier tiers for tasks whose quality gains justify their price. The rapid moves between Fable, Opus, Gemini, and Astra support frequent reassessment.",
  "agent_context": "Open-weight models rose from **7% of gateway tokens in December to 56% in August**, while accounting for 14% of August spend. Average price per token fell **23.2% in August**, and the median high-volume team paid 7.6% less.\n\nTreat model choice as routing, not allegiance: run cheaper or open-weight models where evals permit, and reserve frontier tiers for tasks whose quality gains justify their price. The rapid moves between Fable, Opus, Gemini, and Astra support frequent reassessment.\n\nThis is anonymized traffic from Vercel AI Gateway, not the whole market. Spend uses public list prices, prior months may be revised, and workload shifts are inferred across teams rather than traced token by token.",
  "source": {
    "name": "Vercel",
    "url": "https://vercel.com/blog/ai-gateway-production-index-september-2026",
    "published_at": "2026-09-17T07:00:00.000Z"
  },
  "source_class": "blog_post",
  "content_type": "Engineering Post",
  "layer": "industry",
  "domains": [],
  "topics": [
    "open-models",
    "model-selection",
    "adoption"
  ],
  "verification": {
    "status": "official_source",
    "label": "Official Source",
    "method": "source_feed",
    "verified_at": null
  },
  "uncertainty": [
    "This is anonymized traffic from Vercel AI Gateway, not the whole market. Spend uses public list prices, prior months may be revised, and workload shifts are inferred across teams rather than traced token by token."
  ],
  "connected_context": {
    "meaning": "This strengthens the earlier gateway trend from emerging open-weight adoption to majority token volume, while the much smaller spend share confirms that routing mix can reduce blended cost. It supports frequent, workload-level model reassessment rather than a single-provider default, but remains evidence from one gateway and does not establish comparative task quality.",
    "corpus_size": 807,
    "generated_at": "2026-09-18T10:05:03.869Z",
    "connections": [
      {
        "title": "DeepSeek overtakes Google on volume, cost per token falls 13.6%",
        "source_name": "Vercel",
        "source_url": "https://vercel.com/blog/deepseek-overtakes-google-on-volume-cost-per-token-falls",
        "feed7_url": "https://feed7.dev/p/deepseek-overtakes-google-on-volume-cost-per-token-falls-0xcaifb",
        "reason": "The July data identified open-weight routing as the driver of lower blended token cost; the August figures extend that same gateway trend to 56% of volume and a further price decline."
      },
      {
        "title": "Open-weight models surge to 29% of volume, price per token flattens",
        "source_name": "Vercel",
        "source_url": "https://vercel.com/blog/ai-gateway-production-index-july-2026",
        "feed7_url": "https://feed7.dev/p/ai-gateway-production-index-july-2026-13d6gio",
        "reason": "The earlier 29% open-weight share provides the nearest baseline, making the new 56% figure evidence of a rapid continuation rather than an isolated snapshot."
      },
      {
        "title": "CFOs and the new economics of AI",
        "source_name": "Cursor",
        "source_url": "https://cursor.com/blog/cfo-council",
        "feed7_url": "https://feed7.dev/p/cfo-council-10ctxbn",
        "reason": "Cursor’s multi-model usage and roughly ninefold cost variation reinforce the operational consequence of Vercel’s data: routing choices can materially affect spend and should be revisited by workload."
      },
      {
        "title": "Open Source Is Dead. Long Live Open Source. — Saoud Rizwan, Cline",
        "source_name": "AI Engineer",
        "source_url": "https://www.youtube.com/watch?v=CoEIs6Xm8m8",
        "feed7_url": "https://feed7.dev/p/open-source-is-dead-long-live-open-source-saoud-rizwan-cline-01372wi",
        "reason": "Cline supplies the implementation condition behind cheaper open-weight routing: verification in the harness is needed before lower token cost can be treated as lower cost for completed work."
      }
    ]
  },
  "lifecycle": "Current",
  "published_at": "2026-09-17T07:00:00.000Z",
  "modified_at": "2026-09-17T07:00:00.000Z",
  "supersedes": [],
  "expires_at": null,
  "formats": {
    "html": "https://feed7.dev/p/ai-gateway-production-index-september-2026-0drggwl",
    "json": "https://feed7.dev/p/ai-gateway-production-index-september-2026-0drggwl.json",
    "markdown": "https://feed7.dev/p/ai-gateway-production-index-september-2026-0drggwl.md"
  }
}