{
  "schema_version": 1,
  "as_of": "2026-10-07",
  "currency": "USD",
  "unit": "USD per million tokens",
  "price_basis": "Standard real-time tier list prices from each provider's own pricing page or pricing docs, read on the as_of date. Batch, priority, flex and off-peak prices are in the notes, not in the numbers. null means the page did not state a price (for example 'Contact sales').",
  "fields": {
    "preset_model_id": "The VRAM calculator preset with the same architecture, when the host serves an open model we have a preset for. Hosts may serve a quantized version (noted when they say so).",
    "cached_input_usd_per_mtok": "Price for input tokens served from the provider's prompt cache, where listed."
  },
  "failed_pages": [
    {
      "url": "https://openai.com/api/pricing/",
      "reason": "HTTP 403 (bot protection). Used https://developers.openai.com/api/docs/pricing instead (platform.openai.com/docs/pricing redirects there)."
    },
    {
      "url": "https://groq.com/pricing",
      "reason": "HTTP 308 redirect to the groq.com homepage, which has no prices. Used https://console.groq.com/docs/models (GroqDocs Supported Models, which lists per-model prices)."
    },
    {
      "url": "https://mistral.ai/pricing",
      "reason": "Lists plans only, with no per-model API price table in the raw HTML or via WebFetch; the only API figure is an FAQ example ('Mistral Large costs $0.5/M in, $1.5/M out'). Used https://docs.mistral.ai/inference/pricing instead."
    },
    {
      "url": "https://fireworks.ai/pricing",
      "reason": "No per-token serverless price table (only embeddings, training and GPU prices); it links to https://docs.fireworks.ai/serverless/pricing, which was used."
    }
  ],
  "entries": [
    {
      "id": "openai-gpt-5.4-mini",
      "provider": "OpenAI",
      "model": "gpt-5.4-mini",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.75,
      "output_usd_per_mtok": 4.5,
      "cached_input_usd_per_mtok": 0.075,
      "note": "From the 'All models' Standard table (rows embedded in page data, collapsed by default); row has 3 values mapped as input / cached input / output (no cache-write price). Most recent model actually named '-mini' on the page. Batch/Flex $0.375 / $0.0375 / $2.25; Fast $1.50 / $0.15 / $9.00. No long-context tier shown. Standard tier. Batch and Flex are 50% of Standard. Output price includes reasoning tokens. Regional processing (data residency) and FedRAMP endpoints +10% for models released on/after 2026-03-05. Requested URL platform.openai.com/docs/pricing redirected to developers.openai.com/api/docs/pricing; openai.com/api/pricing/ returned 403.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "openai-gpt-5.4-nano",
      "provider": "OpenAI",
      "model": "gpt-5.4-nano",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.2,
      "output_usd_per_mtok": 1.25,
      "cached_input_usd_per_mtok": 0.02,
      "note": "From the 'All models' Standard table (collapsed by default); 3 values mapped as input / cached input / output. Most recent model actually named '-nano' on the page. Batch/Flex $0.10 / $0.01 / $0.625. No long-context tier shown. Standard tier. Batch and Flex are 50% of Standard. Output price includes reasoning tokens. Regional processing (data residency) and FedRAMP endpoints +10% for models released on/after 2026-03-05. Requested URL platform.openai.com/docs/pricing redirected to developers.openai.com/api/docs/pricing; openai.com/api/pricing/ returned 403.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "openai-gpt-6-astra",
      "provider": "OpenAI",
      "model": "gpt-6-astra",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 10.0,
      "output_usd_per_mtok": 50.0,
      "cached_input_usd_per_mtok": 1.0,
      "note": "Listed first under 'Flagship models'. Short context (<=272K input tokens) shown; long context (>272K input): $20 in / $2 cached / $75 out. Cache writes $12.50 (short) / $25 (long); per page, input tokens are billed as input, cached input, or cache write (not additive). Batch/Flex $5 / $0.50 / $25. Fast $20 / $2 / $100; Ultrafast $60 / $6 / $300. Standard tier. Batch and Flex are 50% of Standard. Output price includes reasoning tokens. Regional processing (data residency) and FedRAMP endpoints +10% for models released on/after 2026-03-05. Requested URL platform.openai.com/docs/pricing redirected to developers.openai.com/api/docs/pricing; openai.com/api/pricing/ returned 403.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "openai-gpt-6-luna",
      "provider": "OpenAI",
      "model": "gpt-6-luna",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.1,
      "output_usd_per_mtok": 0.5,
      "cached_input_usd_per_mtok": 0.01,
      "note": "Flagship-table small tier (current generation uses astra/sol/luna naming rather than mini/nano). Short context (<=272K) shown; long context (>272K): $0.20 in / $0.02 cached / $0.75 out. Cache writes $0.125 (short) / $0.25 (long). Batch/Flex $0.05 / $0.005 / $0.25. Fast $0.20 / $0.02 / $1.00. Standard tier. Batch and Flex are 50% of Standard. Output price includes reasoning tokens. Regional processing (data residency) and FedRAMP endpoints +10% for models released on/after 2026-03-05. Requested URL platform.openai.com/docs/pricing redirected to developers.openai.com/api/docs/pricing; openai.com/api/pricing/ returned 403.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "openai-gpt-6.1-sol",
      "provider": "OpenAI",
      "model": "gpt-6.1-sol",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 2.0,
      "output_usd_per_mtok": 10.0,
      "cached_input_usd_per_mtok": 0.1,
      "note": "Flagship-table mid tier. Short context (<=272K) shown; long context (>272K): $4 in / $0.20 cached / $15 out. Cache writes $2.50 (short) / $5 (long). Batch/Flex $1 / $0.05 / $5. Fast $4 / $0.20 / $20. Also listed: gpt-6-sol at $2 / $0.20 cached / $10. Standard tier. Batch and Flex are 50% of Standard. Output price includes reasoning tokens. Regional processing (data residency) and FedRAMP endpoints +10% for models released on/after 2026-03-05. Requested URL platform.openai.com/docs/pricing redirected to developers.openai.com/api/docs/pricing; openai.com/api/pricing/ returned 403.",
      "source_url": "https://developers.openai.com/api/docs/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "anthropic-claude-fable-5.1",
      "provider": "Anthropic",
      "model": "Claude Fable 5.1",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 10.0,
      "output_usd_per_mtok": 50.0,
      "cached_input_usd_per_mtok": 0.25,
      "note": "Top row of model table ('For demanding reasoning and long-horizon agentic work'). Cache hit is 0.025x input ($0.25). 5m write $12.50, 1h write $20. Batch $5 / $25. Full 1M context at standard pricing (Claude 4.6+). Page says Claude 4.7 and later models use a newer tokenizer that produces about 30% more tokens for the same text, so compare per-task rather than per-token cost. Batch API 50% off input and output. Prompt caching: 5-min write 1.25x input, 1-hour write 2x input. inference_geo 'us' (data residency) 1.1x on all token categories for Claude 4.6+ models. Requested docs.claude.com URL redirected to platform.claude.com.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "anthropic-claude-haiku-4.5",
      "provider": "Anthropic",
      "model": "Claude Haiku 4.5",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 1.0,
      "output_usd_per_mtok": 5.0,
      "cached_input_usd_per_mtok": 0.1,
      "note": "Current Haiku. Cache hit $0.10 (0.1x). 5m write $1.25, 1h write $2. Batch $0.50 / $2.50. Uses the previous tokenizer (Claude 4.6 and earlier). Batch API 50% off input and output. Prompt caching: 5-min write 1.25x input, 1-hour write 2x input. inference_geo 'us' (data residency) 1.1x on all token categories for Claude 4.6+ models. Requested docs.claude.com URL redirected to platform.claude.com.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "anthropic-claude-opus-5.5",
      "provider": "Anthropic",
      "model": "Claude Opus 5.5",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 4.0,
      "output_usd_per_mtok": 20.0,
      "cached_input_usd_per_mtok": 0.2,
      "note": "Current Opus. Cache hit is 0.05x input ($0.20). 5m write $5, 1h write $8. Batch $2 / $10. Fast mode (research preview) $8 in / $40 out. Full 1M context at standard pricing (Claude 4.6+). Page says Claude 4.7 and later models use a newer tokenizer that produces about 30% more tokens for the same text, so compare per-task rather than per-token cost. Batch API 50% off input and output. Prompt caching: 5-min write 1.25x input, 1-hour write 2x input. inference_geo 'us' (data residency) 1.1x on all token categories for Claude 4.6+ models. Requested docs.claude.com URL redirected to platform.claude.com.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "anthropic-claude-sonnet-5.5",
      "provider": "Anthropic",
      "model": "Claude Sonnet 5.5",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 2.0,
      "output_usd_per_mtok": 10.0,
      "cached_input_usd_per_mtok": 0.2,
      "note": "Current Sonnet. Cache hit $0.20 (0.1x). 5m write $2.50, 1h write $4. Batch $1 / $5. Full 1M context at standard pricing (Claude 4.6+). Page says Claude 4.7 and later models use a newer tokenizer that produces about 30% more tokens for the same text, so compare per-task rather than per-token cost. Batch API 50% off input and output. Prompt caching: 5-min write 1.25x input, 1-hour write 2x input. inference_geo 'us' (data residency) 1.1x on all token categories for Claude 4.6+ models. Requested docs.claude.com URL redirected to platform.claude.com.",
      "source_url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "google-gemini-2.5-pro",
      "provider": "Google",
      "model": "Gemini 2.5 Pro (gemini-2.5-pro)",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 1.25,
      "output_usd_per_mtok": 10.0,
      "cached_input_usd_per_mtok": 0.125,
      "note": "Older generally available Pro model, kept for reference. Prompts <=200k shown; >200k: $2.50 in / $15.00 out / $0.25 cached. Batch $0.625 / $5.00 (<=200k). Paid tier, Standard. Output price includes thinking tokens. Batch and Flex are 50% of Standard. First plain curl was redirected to a Google sign-in check; the page loaded fine with cookies kept (public, no login).",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing?hl=en",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "google-gemini-3.1-pro-preview",
      "provider": "Google",
      "model": "Gemini 3.1 Pro Preview (gemini-3.1-pro-preview)",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 2.0,
      "output_usd_per_mtok": 12.0,
      "cached_input_usd_per_mtok": 0.2,
      "note": "Only Gemini 3.x Pro text model with its own price table (still 'Preview'). Prompts <=200k tokens shown; prompts >200k: $4.00 in / $18.00 out / $0.40 cached. Context-cache storage $4.50 per 1M tokens per hour. Batch/Flex $1.00 / $6.00 (<=200k), $2.00 / $9.00 (>200k). Priority $3.60 / $21.60 (<=200k). Paid tier, Standard. Output price includes thinking tokens. Batch and Flex are 50% of Standard. First plain curl was redirected to a Google sign-in check; the page loaded fine with cookies kept (public, no login).",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing?hl=en",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "google-gemini-3.5-flash",
      "provider": "Google",
      "model": "Gemini 3.5 Flash (gemini-3.5-flash)",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 1.5,
      "output_usd_per_mtok": 9.0,
      "cached_input_usd_per_mtok": 0.15,
      "note": "Page calls it 'Our earlier Flash model'; it is not on the promotional pricing the 3.6 to 3.8 Flash models have. Cache storage $1.00 per 1M tokens per hour. Batch/Flex $0.75 / $4.50. Paid tier, Standard. Output price includes thinking tokens. Batch and Flex are 50% of Standard. First plain curl was redirected to a Google sign-in check; the page loaded fine with cookies kept (public, no login).",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing?hl=en",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "google-gemini-3.5-flash-lite",
      "provider": "Google",
      "model": "Gemini 3.5 Flash-Lite (gemini-3.5-flash-lite)",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 2.5,
      "cached_input_usd_per_mtok": 0.03,
      "note": "Input price applies to text/image/video/audio. Cache storage $1.00 per 1M tokens per hour. Batch $0.15 / $1.25. Older Gemini 3.1 Flash-Lite is cheaper: $0.25 in (text/image/video; $0.50 audio) / $1.50 out / $0.025 cached. Paid tier, Standard. Output price includes thinking tokens. Batch and Flex are 50% of Standard. First plain curl was redirected to a Google sign-in check; the page loaded fine with cookies kept (public, no login).",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing?hl=en",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "google-gemini-3.8-flash",
      "provider": "Google",
      "model": "Gemini 3.8 Flash (gemini-3.8-flash)",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.75,
      "output_usd_per_mtok": 3.75,
      "cached_input_usd_per_mtok": 0.075,
      "note": "Page calls it 'Our most intelligent Flash model'. PROMOTIONAL: $0.75 in / $3.75 out / $0.075 cached through 2026-12-31; from 2027-01-01 $1.50 / $7.50 / $0.15. Cache storage $0.50 per 1M tokens per hour (rises to $1.00). No >200k tier shown. Batch/Flex $0.375 / $1.875. Priority $1.35 / $6.75. Gemini 3.7 Flash and 3.6 Flash carry the same promotional prices. Paid tier, Standard. Output price includes thinking tokens. Batch and Flex are 50% of Standard. First plain curl was redirected to a Google sign-in check; the page loaded fine with cookies kept (public, no login).",
      "source_url": "https://ai.google.dev/gemini-api/docs/pricing?hl=en",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepseek-flash",
      "provider": "DeepSeek",
      "model": "deepseek-flash (DeepSeek-V4.1-Flash)",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4.1-Flash",
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": 0.006,
      "note": "Cache-miss input $0.30, cache-hit input $0.006, output $1.20 (peak). Off-peak: $0.15 / $0.003 / $0.60. Legacy names deepseek-v4-flash and deepseek-v4-flash-vision-exp are served by V4.1-Flash at this price. Weights on HF (MIT) per HF API check. PEAK rates recorded (the higher, undiscounted rate). Off-peak is half price: peak hours are 01:00-04:00 and 06:00-10:00 UTC, Mon-Fri, excluding Chinese public holidays; all other hours (most of the week) are off-peak. Context 1M, max output 384K. Thinking mode is on by default.",
      "source_url": "https://api-docs.deepseek.com/quick_start/pricing/",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepseek-v4-pro",
      "provider": "DeepSeek",
      "model": "deepseek-v4-pro (DeepSeek-V4-Pro-0813)",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4-Pro-0813",
      "input_usd_per_mtok": 1.32,
      "output_usd_per_mtok": 3.96,
      "cached_input_usd_per_mtok": 0.044,
      "note": "Cache-miss input $1.32, cache-hit input $0.044, output $3.96 (peak). Off-peak: $0.66 / $0.022 / $1.98. No vision. Weights on HF (MIT) per HF API check. PEAK rates recorded (the higher, undiscounted rate). Off-peak is half price: peak hours are 01:00-04:00 and 06:00-10:00 UTC, Mon-Fri, excluding Chinese public holidays; all other hours (most of the week) are off-peak. Context 1M, max output 384K. Thinking mode is on by default.",
      "source_url": "https://api-docs.deepseek.com/quick_start/pricing/",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "mistral-large-3",
      "provider": "Mistral",
      "model": "Mistral Large 3",
      "open_weights": true,
      "open_model_hint": "mistralai/Mistral-Large-3-675B-Instruct-2512",
      "input_usd_per_mtok": 0.5,
      "output_usd_per_mtok": 1.5,
      "cached_input_usd_per_mtok": 0.05,
      "note": "675B MoE with open weights (Apache-2.0 on HF). The mistral.ai/pricing FAQ example also quotes 'Mistral Large' at $0.5 in / $1.5 out. Standard tier, default (non-regional) view of docs.mistral.ai/inference/pricing. mistral.ai/pricing (the requested URL) shows only plans, no per-model API table. Batch is 50% off per Mistral's pages.",
      "source_url": "https://docs.mistral.ai/inference/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "mistral-large-4",
      "provider": "Mistral",
      "model": "Mistral Large 4",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.68,
      "output_usd_per_mtok": 2.09,
      "cached_input_usd_per_mtok": 0.07,
      "note": "SALE PRICE recorded (what is charged today). The page shows the original price struck through: $1.36 in / $0.14 cached / $4.18 out. No sale end date is given. No mistralai Large 4 repo found on Hugging Face, so treated as closed weights. Standard tier, default (non-regional) view of docs.mistral.ai/inference/pricing. mistral.ai/pricing (the requested URL) shows only plans, no per-model API table. Batch is 50% off per Mistral's pages.",
      "source_url": "https://docs.mistral.ai/inference/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "mistral-medium-3.5",
      "provider": "Mistral",
      "model": "Mistral Medium 3.5",
      "open_weights": true,
      "open_model_hint": "mistralai/Mistral-Medium-3.5-128B",
      "input_usd_per_mtok": 1.5,
      "output_usd_per_mtok": 7.5,
      "cached_input_usd_per_mtok": 0.15,
      "note": "The pricing page recommends it for most tasks and coding. HF has mistralai/Mistral-Medium-3.5-128B under a non-Apache license ('other', probably Mistral Research License), so check commercial self-hosting terms. Standard tier, default (non-regional) view of docs.mistral.ai/inference/pricing. mistral.ai/pricing (the requested URL) shows only plans, no per-model API table. Batch is 50% off per Mistral's pages.",
      "source_url": "https://docs.mistral.ai/inference/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "mistral-small-4",
      "provider": "Mistral",
      "model": "Mistral Small 4",
      "open_weights": true,
      "open_model_hint": "mistralai/Mistral-Small-4-119B-2603",
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.6,
      "cached_input_usd_per_mtok": 0.015,
      "note": "Open weights (Apache-2.0 on HF, 119B). Same page lists Ministral 3 14B $0.2/$0.2, 8B $0.15/$0.15, 3B $0.1/$0.1, and third-party Z.ai GLM 5.3 at $1.4 / $0.14 cached / $4.4. Standard tier, default (non-regional) view of docs.mistral.ai/inference/pricing. mistral.ai/pricing (the requested URL) shows only plans, no per-model API table. Batch is 50% off per Mistral's pages.",
      "source_url": "https://docs.mistral.ai/inference/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-deepseek-v4-flash-0731",
      "provider": "Together AI",
      "model": "DeepSeek V4 Flash 0731",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4-Flash-0731",
      "input_usd_per_mtok": 0.14,
      "output_usd_per_mtok": 0.28,
      "cached_input_usd_per_mtok": 0.03,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-deepseek-v4-pro-0813",
      "provider": "Together AI",
      "model": "DeepSeek V4 Pro 0813",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4-Pro-0813",
      "input_usd_per_mtok": 1.32,
      "output_usd_per_mtok": 3.96,
      "cached_input_usd_per_mtok": 0.13,
      "note": "Input and output match DeepSeek's own peak prices; cached is $0.13 here vs $0.044 at DeepSeek. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-deepseek-v4.1-flash",
      "provider": "Together AI",
      "model": "DeepSeek V4.1 Flash",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4.1-Flash",
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": 0.006,
      "note": "Matches DeepSeek's own peak prices. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-gemma-4-31b",
      "provider": "Together AI",
      "model": "Gemma 4 31B",
      "open_weights": true,
      "open_model_hint": "google/gemma-4-31B-it",
      "input_usd_per_mtok": 0.39,
      "output_usd_per_mtok": 0.97,
      "cached_input_usd_per_mtok": null,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gemma-4-31b"
    },
    {
      "id": "together-glm-5.3",
      "provider": "Together AI",
      "model": "GLM-5.3",
      "open_weights": true,
      "open_model_hint": "zai-org/GLM-5.3",
      "input_usd_per_mtok": 1.4,
      "output_usd_per_mtok": 4.4,
      "cached_input_usd_per_mtok": 0.26,
      "note": "GLM-5.2 is also listed at the same $1.40 / $0.26 / $4.40. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "glm-5.3"
    },
    {
      "id": "together-glm-5.3-flash",
      "provider": "Together AI",
      "model": "GLM-5.3-Flash",
      "open_weights": true,
      "open_model_hint": "zai-org/GLM-5.3-Flash",
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.5,
      "cached_input_usd_per_mtok": 0.03,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-gpt-oss-120b",
      "provider": "Together AI",
      "model": "gpt-oss-120B",
      "open_weights": true,
      "open_model_hint": "openai/gpt-oss-120b",
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.6,
      "cached_input_usd_per_mtok": null,
      "note": "No cached price shown. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gpt-oss-120b"
    },
    {
      "id": "together-kimi-k3",
      "provider": "Together AI",
      "model": "Kimi K3",
      "open_weights": true,
      "open_model_hint": "moonshotai/Kimi-K3",
      "input_usd_per_mtok": 2.7,
      "output_usd_per_mtok": 13.5,
      "cached_input_usd_per_mtok": 0.27,
      "note": "Marked 'PROMO' in the Vision tab of the same table, so this price may be temporary. Kimi K2.x is not listed on Together's page. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-llama-3-8b-instruct-lite",
      "provider": "Together AI",
      "model": "Llama 3 8B Instruct Lite",
      "open_weights": true,
      "open_model_hint": "meta-llama/Meta-Llama-3-8B-Instruct",
      "input_usd_per_mtok": 0.14,
      "output_usd_per_mtok": 0.14,
      "cached_input_usd_per_mtok": null,
      "note": "This is Llama 3 (not 3.1) 8B 'Lite'. Together lists no Llama 3.1 8B chat model. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-llama-3.3-70b",
      "provider": "Together AI",
      "model": "Llama 3.3 70B",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.3-70B-Instruct",
      "input_usd_per_mtok": 1.04,
      "output_usd_per_mtok": 1.04,
      "cached_input_usd_per_mtok": null,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.3-70b"
    },
    {
      "id": "together-minimax-m2.7",
      "provider": "Together AI",
      "model": "MiniMax M2.7",
      "open_weights": true,
      "open_model_hint": "MiniMaxAI/MiniMax-M2.7",
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": 0.06,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "minimax-m2.7"
    },
    {
      "id": "together-minimax-m3",
      "provider": "Together AI",
      "model": "MiniMax M3",
      "open_weights": true,
      "open_model_hint": "MiniMaxAI/MiniMax-M3",
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": 0.06,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-qwen3-235b-a22b-instruct-2507-fp8",
      "provider": "Together AI",
      "model": "Qwen3 235B A22B Instruct 2507 FP8 Throughput",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3-235B-A22B-Instruct-2507-FP8",
      "input_usd_per_mtok": 0.2,
      "output_usd_per_mtok": 0.6,
      "cached_input_usd_per_mtok": null,
      "note": "FP8, 'Throughput' variant (as named). Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3-235b-a22b"
    },
    {
      "id": "together-qwen3.5-397b-a17b",
      "provider": "Together AI",
      "model": "Qwen3.5-397B-A17B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.5-397B-A17B",
      "input_usd_per_mtok": 0.6,
      "output_usd_per_mtok": 3.6,
      "cached_input_usd_per_mtok": 0.35,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-qwen3.5-9b",
      "provider": "Together AI",
      "model": "Qwen3.5 9B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.5-9B",
      "input_usd_per_mtok": 0.17,
      "output_usd_per_mtok": 0.25,
      "cached_input_usd_per_mtok": null,
      "note": "Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-qwen3.6-plus",
      "provider": "Together AI",
      "model": "Qwen3.6-Plus",
      "open_weights": false,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.5,
      "output_usd_per_mtok": 3.0,
      "cached_input_usd_per_mtok": null,
      "note": "No Qwen/Qwen3.6-Plus repo on HF (only Qwen3.6-27B and 35B-A3B), so treated as Alibaba's closed 'Plus' API model resold by Together. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-qwen3.8-2.4t-a95b",
      "provider": "Together AI",
      "model": "Qwen3.8-2.4T-A95B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.8-2.4T-A95B",
      "input_usd_per_mtok": 2.0,
      "output_usd_per_mtok": 6.0,
      "cached_input_usd_per_mtok": 0.25,
      "note": "Page URL slug is 'qwen3-8-max', so this is probably what Fireworks and DeepInfra call 'Qwen3.8 Max'. HF license is 'other'. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "together-qwen3.8-flash",
      "provider": "Together AI",
      "model": "Qwen3.8 Flash",
      "open_weights": null,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.47,
      "cached_input_usd_per_mtok": null,
      "note": "Open-weights status unclear: HF has Qwen/Qwen3.8-Flash-Next but no 'Qwen3.8-Flash' repo. Serverless Chat table. The page has a 'Batch API price' toggle, but batch prices are not in the static HTML (filled in by JavaScript). Quantization not stated unless named.",
      "source_url": "https://www.together.ai/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-bucket-dense-4b-16b",
      "provider": "Fireworks AI",
      "model": "Other base models: 4B to 16B parameters",
      "open_weights": true,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.2,
      "output_usd_per_mtok": 0.2,
      "cached_input_usd_per_mtok": null,
      "note": "Would cover Llama 3.1 8B, Qwen3 8B, Qwen3 14B and Gemma 3 12B if served. SIZE-BUCKET RULE: 'For any text or vision model not listed individually, pricing is set by parameter count and architecture.' The same price applies to input and output, with no separate cached-input price. This page does NOT confirm which specific models are on serverless; check Fireworks' model library before mapping a model to this bucket. Batch is 50%.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-bucket-dense-over-16b",
      "provider": "Fireworks AI",
      "model": "Other base models: more than 16B parameters",
      "open_weights": true,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.9,
      "output_usd_per_mtok": 0.9,
      "cached_input_usd_per_mtok": null,
      "note": "Would cover dense Llama 3.3 70B, Llama 3.1 405B, Qwen3 32B, Gemma 3 27B and Mistral Small 24B if served. SIZE-BUCKET RULE: 'For any text or vision model not listed individually, pricing is set by parameter count and architecture.' The same price applies to input and output, with no separate cached-input price. This page does NOT confirm which specific models are on serverless; check Fireworks' model library before mapping a model to this bucket. Batch is 50%.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-bucket-dense-under-4b",
      "provider": "Fireworks AI",
      "model": "Other base models: less than 4B parameters",
      "open_weights": true,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.1,
      "output_usd_per_mtok": 0.1,
      "cached_input_usd_per_mtok": null,
      "note": "Applies to dense models under 4B parameters. SIZE-BUCKET RULE: 'For any text or vision model not listed individually, pricing is set by parameter count and architecture.' The same price applies to input and output, with no separate cached-input price. This page does NOT confirm which specific models are on serverless; check Fireworks' model library before mapping a model to this bucket. Batch is 50%.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-bucket-moe-56b-176b",
      "provider": "Fireworks AI",
      "model": "Other base models: MoE 56.1B to 176B parameters (e.g. DBRX, Mixtral 8x22B)",
      "open_weights": true,
      "open_model_hint": null,
      "input_usd_per_mtok": 1.2,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": null,
      "note": "No bucket is listed for MoE above 176B (e.g. Qwen3 235B-A22B), so those models have no rule-based price. SIZE-BUCKET RULE: 'For any text or vision model not listed individually, pricing is set by parameter count and architecture.' The same price applies to input and output, with no separate cached-input price. This page does NOT confirm which specific models are on serverless; check Fireworks' model library before mapping a model to this bucket. Batch is 50%.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-bucket-moe-up-to-56b",
      "provider": "Fireworks AI",
      "model": "Other base models: MoE up to 56B parameters (e.g. Mixtral 8x7B)",
      "open_weights": true,
      "open_model_hint": null,
      "input_usd_per_mtok": 0.5,
      "output_usd_per_mtok": 0.5,
      "cached_input_usd_per_mtok": null,
      "note": "Would cover Qwen3 30B-A3B and gpt-oss-20b (about 21B MoE) if served. SIZE-BUCKET RULE: 'For any text or vision model not listed individually, pricing is set by parameter count and architecture.' The same price applies to input and output, with no separate cached-input price. This page does NOT confirm which specific models are on serverless; check Fireworks' model library before mapping a model to this bucket. Batch is 50%.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-deepseek-v4.1-flash",
      "provider": "Fireworks AI",
      "model": "DeepSeek V4.1 Flash",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4.1-Flash",
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": 0.006,
      "note": "US region $0.45 / $0.009 / $1.80. Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-glm-5.3",
      "provider": "Fireworks AI",
      "model": "GLM 5.3",
      "open_weights": true,
      "open_model_hint": "zai-org/GLM-5.3",
      "input_usd_per_mtok": 1.4,
      "output_usd_per_mtok": 4.4,
      "cached_input_usd_per_mtok": 0.26,
      "note": "GLM 5.3 Fast and US region both $2.10 / $0.39 / $6.60. Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "glm-5.3"
    },
    {
      "id": "fireworks-glm-5.3-flash",
      "provider": "Fireworks AI",
      "model": "GLM 5.3 Flash",
      "open_weights": true,
      "open_model_hint": "zai-org/GLM-5.3-Flash",
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.5,
      "cached_input_usd_per_mtok": 0.03,
      "note": "Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-gpt-oss-120b",
      "provider": "Fireworks AI",
      "model": "OpenAI GPT OSS 120B",
      "open_weights": true,
      "open_model_hint": "openai/gpt-oss-120b",
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.6,
      "cached_input_usd_per_mtok": 0.015,
      "note": "Priority $0.18 / $0.018 / $0.72. Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gpt-oss-120b"
    },
    {
      "id": "fireworks-kimi-k3",
      "provider": "Fireworks AI",
      "model": "Kimi K3",
      "open_weights": true,
      "open_model_hint": "moonshotai/Kimi-K3",
      "input_usd_per_mtok": 3.0,
      "output_usd_per_mtok": 15.0,
      "cached_input_usd_per_mtok": 0.3,
      "note": "Kimi K3 Fast $4.50 / $0.45 / $22.50; US region $4.50 / $0.45 / $22.50. Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-minimax-m3",
      "provider": "Fireworks AI",
      "model": "MiniMax M3",
      "open_weights": true,
      "open_model_hint": "MiniMaxAI/MiniMax-M3",
      "input_usd_per_mtok": 0.3,
      "output_usd_per_mtok": 1.2,
      "cached_input_usd_per_mtok": 0.06,
      "note": "Priority $0.45 / $0.09 / $1.80. Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "fireworks-qwen-3.8-max",
      "provider": "Fireworks AI",
      "model": "Qwen 3.8 Max",
      "open_weights": null,
      "open_model_hint": null,
      "input_usd_per_mtok": 2.0,
      "output_usd_per_mtok": 6.0,
      "cached_input_usd_per_mtok": 0.25,
      "note": "Same price as Together's 'Qwen3.8-2.4T-A95B' (slug qwen3-8-max), so probably the open Qwen/Qwen3.8-2.4T-A95B, but this page does not say so. Priority $3.00 / $0.375 / $9.00. Standard serverless tier; cells are input / cached input / output. Batch inference is 50% of serverless on input and output. Priority costs about 1.25x (1.5x for some models); '(US)' region variants cost 1.5x. fireworks.ai/pricing has no per-token table and links to this docs page.",
      "source_url": "https://docs.fireworks.ai/serverless/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-deepseek-v3.2",
      "provider": "DeepInfra",
      "model": "DeepSeek-V3.2",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V3.2",
      "input_usd_per_mtok": 0.26,
      "output_usd_per_mtok": 0.38,
      "cached_input_usd_per_mtok": 0.13,
      "note": "Context 160k. DeepSeek-V3.1 is also listed: $0.25 / $0.13 cached / $0.95. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "deepseek-v3.2"
    },
    {
      "id": "deepinfra-deepseek-v4-flash",
      "provider": "DeepInfra",
      "model": "DeepSeek-V4-Flash",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4-Flash",
      "input_usd_per_mtok": 0.09,
      "output_usd_per_mtok": 0.18,
      "cached_input_usd_per_mtok": 0.018,
      "note": "Context 1024k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-deepseek-v4-flash-0731",
      "provider": "DeepInfra",
      "model": "DeepSeek-V4-Flash-0731",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4-Flash-0731",
      "input_usd_per_mtok": 0.06,
      "output_usd_per_mtok": 0.18,
      "cached_input_usd_per_mtok": 0.015,
      "note": "Context 1024k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-deepseek-v4-pro",
      "provider": "DeepInfra",
      "model": "DeepSeek-V4-Pro",
      "open_weights": true,
      "open_model_hint": "deepseek-ai/DeepSeek-V4-Pro",
      "input_usd_per_mtok": 1.3,
      "output_usd_per_mtok": 2.6,
      "cached_input_usd_per_mtok": 0.1,
      "note": "Context 1024k. This is the original V4-Pro checkpoint; DeepSeek's own API now serves V4-Pro-0813. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-gemma-3-27b-it",
      "provider": "DeepInfra",
      "model": "gemma-3-27b-it",
      "open_weights": true,
      "open_model_hint": "google/gemma-3-27b-it",
      "input_usd_per_mtok": 0.08,
      "output_usd_per_mtok": 0.16,
      "cached_input_usd_per_mtok": null,
      "note": "Context 128k. gemma-3-12b-it is also listed at $0.05 / $0.15. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-gemma-4-26b-a4b-it",
      "provider": "DeepInfra",
      "model": "gemma-4-26B-A4B-it",
      "open_weights": true,
      "open_model_hint": "google/gemma-4-26B-A4B-it",
      "input_usd_per_mtok": 0.07,
      "output_usd_per_mtok": 0.34,
      "cached_input_usd_per_mtok": null,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gemma-4-26b-a4b"
    },
    {
      "id": "deepinfra-gemma-4-31b-it",
      "provider": "DeepInfra",
      "model": "gemma-4-31B-it",
      "open_weights": true,
      "open_model_hint": "google/gemma-4-31B-it",
      "input_usd_per_mtok": 0.2,
      "output_usd_per_mtok": 0.4,
      "cached_input_usd_per_mtok": null,
      "note": "Context 256k. Variants also listed: gemma-4-31B-it-turbo $0.09 / $0.05 cached / $0.34, and gemma-4-31B-it-Ultra $0.27 / $0.76 (128k). Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gemma-4-31b"
    },
    {
      "id": "deepinfra-kimi-k2.6",
      "provider": "DeepInfra",
      "model": "Kimi-K2.6",
      "open_weights": true,
      "open_model_hint": "moonshotai/Kimi-K2.6",
      "input_usd_per_mtok": 0.75,
      "output_usd_per_mtok": 3.5,
      "cached_input_usd_per_mtok": 0.15,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-kimi-k3",
      "provider": "DeepInfra",
      "model": "Kimi-K3",
      "open_weights": true,
      "open_model_hint": "moonshotai/Kimi-K3",
      "input_usd_per_mtok": 2.85,
      "output_usd_per_mtok": 14.25,
      "cached_input_usd_per_mtok": 0.285,
      "note": "Context 1024k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-llama-3.1-8b-instruct-turbo",
      "provider": "DeepInfra",
      "model": "Meta-Llama-3.1-8B-Instruct-Turbo",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.1-8B-Instruct",
      "input_usd_per_mtok": 0.02,
      "output_usd_per_mtok": 0.04,
      "cached_input_usd_per_mtok": null,
      "note": "Turbo variant; context 128k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.1-8b"
    },
    {
      "id": "deepinfra-llama-3.3-70b-instruct-turbo",
      "provider": "DeepInfra",
      "model": "Llama-3.3-70B-Instruct-Turbo",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.3-70B-Instruct",
      "input_usd_per_mtok": 0.1,
      "output_usd_per_mtok": 0.32,
      "cached_input_usd_per_mtok": null,
      "note": "Turbo variant; context 128k. Meta-Llama-3.1-70B-Instruct-Turbo is also listed at $0.40 / $0.40. No 405B listed. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.3-70b"
    },
    {
      "id": "deepinfra-mistral-small-3.2-24b",
      "provider": "DeepInfra",
      "model": "Mistral-Small-3.2-24B-Instruct-2506",
      "open_weights": true,
      "open_model_hint": "mistralai/Mistral-Small-3.2-24B-Instruct-2506",
      "input_usd_per_mtok": 0.075,
      "output_usd_per_mtok": 0.2,
      "cached_input_usd_per_mtok": null,
      "note": "Context 125k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "mistral-small-3.2-24b"
    },
    {
      "id": "deepinfra-qwen3-14b",
      "provider": "DeepInfra",
      "model": "Qwen3-14B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3-14B",
      "input_usd_per_mtok": 0.12,
      "output_usd_per_mtok": 0.24,
      "cached_input_usd_per_mtok": null,
      "note": "Context 40k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3-14b"
    },
    {
      "id": "deepinfra-qwen3-235b-a22b-instruct-2507",
      "provider": "DeepInfra",
      "model": "Qwen3-235B-A22B-Instruct-2507",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3-235B-A22B-Instruct-2507",
      "input_usd_per_mtok": 0.09,
      "output_usd_per_mtok": 0.55,
      "cached_input_usd_per_mtok": null,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3-235b-a22b"
    },
    {
      "id": "deepinfra-qwen3-30b-a3b",
      "provider": "DeepInfra",
      "model": "Qwen3-30B-A3B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3-30B-A3B",
      "input_usd_per_mtok": 0.12,
      "output_usd_per_mtok": 0.5,
      "cached_input_usd_per_mtok": null,
      "note": "Context 40k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3-30b-a3b"
    },
    {
      "id": "deepinfra-qwen3-32b",
      "provider": "DeepInfra",
      "model": "Qwen3-32B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3-32B",
      "input_usd_per_mtok": 0.08,
      "output_usd_per_mtok": 0.28,
      "cached_input_usd_per_mtok": null,
      "note": "Context 40k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3-32b"
    },
    {
      "id": "deepinfra-qwen3.5-27b",
      "provider": "DeepInfra",
      "model": "Qwen3.5-27B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.5-27B",
      "input_usd_per_mtok": 0.26,
      "output_usd_per_mtok": 2.6,
      "cached_input_usd_per_mtok": null,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-qwen3.5-35b-a3b",
      "provider": "DeepInfra",
      "model": "Qwen3.5-35B-A3B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.5-35B-A3B",
      "input_usd_per_mtok": 0.14,
      "output_usd_per_mtok": 1.0,
      "cached_input_usd_per_mtok": 0.05,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-qwen3.5-397b-a17b",
      "provider": "DeepInfra",
      "model": "Qwen3.5-397B-A17B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.5-397B-A17B",
      "input_usd_per_mtok": 0.45,
      "output_usd_per_mtok": 3.0,
      "cached_input_usd_per_mtok": 0.22,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-qwen3.5-9b",
      "provider": "DeepInfra",
      "model": "Qwen3.5-9B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.5-9B",
      "input_usd_per_mtok": 0.1,
      "output_usd_per_mtok": 0.15,
      "cached_input_usd_per_mtok": null,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-qwen3.6-27b",
      "provider": "DeepInfra",
      "model": "Qwen3.6-27B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.6-27B",
      "input_usd_per_mtok": 0.32,
      "output_usd_per_mtok": 3.2,
      "cached_input_usd_per_mtok": null,
      "note": "Context 256k. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "deepinfra-qwen3.6-35b-a3b",
      "provider": "DeepInfra",
      "model": "Qwen3.6-35B-A3B",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.6-35B-A3B",
      "input_usd_per_mtok": 0.1,
      "output_usd_per_mtok": 0.95,
      "cached_input_usd_per_mtok": 0.1,
      "note": "Context 256k. The page shows cached input equal to input ($0.10 / $0.10 cached). Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3.6-35b-a3b"
    },
    {
      "id": "deepinfra-qwen3.8-max",
      "provider": "DeepInfra",
      "model": "Qwen3.8-Max",
      "open_weights": null,
      "open_model_hint": null,
      "input_usd_per_mtok": 1.65,
      "output_usd_per_mtok": 4.951,
      "cached_input_usd_per_mtok": 0.206,
      "note": "Context 250k. Probably the same model as Together's Qwen3.8-2.4T-A95B, but there is no Qwen/Qwen3.8-Max repo on HF and DeepInfra also resells closed models (e.g. Qwen3-Max, Gemini, Claude), so open-weights status is unconfirmed. Standard service tier (1x). Priority 1.5x, Flex 0.8x. Hint is taken from DeepInfra's model URL path. Quantization is not stated on the pricing page; DeepInfra 'Turbo' variants are often quantized, so check the model page.",
      "source_url": "https://deepinfra.com/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": null
    },
    {
      "id": "groq-gpt-oss-120b",
      "provider": "Groq",
      "model": "GPT OSS 120B (openai/gpt-oss-120b)",
      "open_weights": true,
      "open_model_hint": "openai/gpt-oss-120b",
      "input_usd_per_mtok": 0.15,
      "output_usd_per_mtok": 0.6,
      "cached_input_usd_per_mtok": null,
      "note": "Production model, about 500 tokens/s. groq.com/pricing now redirects (308) to the homepage, so prices come from GroqDocs 'Supported Models' (console.groq.com/docs/models), the provider's own public page. No cached-input price shown on that page.",
      "source_url": "https://console.groq.com/docs/models",
      "as_of": "2026-10-07",
      "preset_model_id": "gpt-oss-120b"
    },
    {
      "id": "groq-gpt-oss-20b",
      "provider": "Groq",
      "model": "GPT OSS 20B (openai/gpt-oss-20b)",
      "open_weights": true,
      "open_model_hint": "openai/gpt-oss-20b",
      "input_usd_per_mtok": 0.075,
      "output_usd_per_mtok": 0.3,
      "cached_input_usd_per_mtok": null,
      "note": "Production model, about 1000 tokens/s. groq.com/pricing now redirects (308) to the homepage, so prices come from GroqDocs 'Supported Models' (console.groq.com/docs/models), the provider's own public page. No cached-input price shown on that page.",
      "source_url": "https://console.groq.com/docs/models",
      "as_of": "2026-10-07",
      "preset_model_id": "gpt-oss-20b"
    },
    {
      "id": "groq-llama-3.1-8b-instant",
      "provider": "Groq",
      "model": "Llama 3.1 8B (llama-3.1-8b-instant)",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.1-8B-Instruct",
      "input_usd_per_mtok": null,
      "output_usd_per_mtok": null,
      "cached_input_usd_per_mtok": null,
      "note": "Price shown as 'Contact Sales' (tagged Enterprise); no public per-token price. groq.com/pricing now redirects (308) to the homepage, so prices come from GroqDocs 'Supported Models' (console.groq.com/docs/models), the provider's own public page. No cached-input price shown on that page.",
      "source_url": "https://console.groq.com/docs/models",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.1-8b"
    },
    {
      "id": "groq-llama-3.3-70b-versatile",
      "provider": "Groq",
      "model": "Llama 3.3 70B (llama-3.3-70b-versatile)",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.3-70B-Instruct",
      "input_usd_per_mtok": null,
      "output_usd_per_mtok": null,
      "cached_input_usd_per_mtok": null,
      "note": "Price shown as 'Contact Sales' (tagged Enterprise); no public per-token price. groq.com/pricing now redirects (308) to the homepage, so prices come from GroqDocs 'Supported Models' (console.groq.com/docs/models), the provider's own public page. No cached-input price shown on that page.",
      "source_url": "https://console.groq.com/docs/models",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.3-70b"
    },
    {
      "id": "groq-minimax-m2.7",
      "provider": "Groq",
      "model": "MiniMax M2.7 (minimaxai/minimax-m2.7)",
      "open_weights": true,
      "open_model_hint": "MiniMaxAI/MiniMax-M2.7",
      "input_usd_per_mtok": null,
      "output_usd_per_mtok": null,
      "cached_input_usd_per_mtok": null,
      "note": "Preview, tagged Enterprise; price shown as 'Contact Sales'. groq.com/pricing now redirects (308) to the homepage, so prices come from GroqDocs 'Supported Models' (console.groq.com/docs/models), the provider's own public page. No cached-input price shown on that page.",
      "source_url": "https://console.groq.com/docs/models",
      "as_of": "2026-10-07",
      "preset_model_id": "minimax-m2.7"
    },
    {
      "id": "groq-qwen3.8-27b",
      "provider": "Groq",
      "model": "Qwen/Qwen3.8-27B (qwen/qwen3.8-27b)",
      "open_weights": true,
      "open_model_hint": "Qwen/Qwen3.8-27B",
      "input_usd_per_mtok": 0.8,
      "output_usd_per_mtok": 4.0,
      "cached_input_usd_per_mtok": null,
      "note": "PREVIEW model (evaluation only, may be discontinued at short notice). groq.com/pricing now redirects (308) to the homepage, so prices come from GroqDocs 'Supported Models' (console.groq.com/docs/models), the provider's own public page. No cached-input price shown on that page.",
      "source_url": "https://console.groq.com/docs/models",
      "as_of": "2026-10-07",
      "preset_model_id": "qwen3.8-27b"
    },
    {
      "id": "hyperstack-gpt-oss-120b",
      "provider": "Hyperstack",
      "model": "OpenAI gpt-oss-120b",
      "open_weights": true,
      "open_model_hint": "openai/gpt-oss-120b",
      "input_usd_per_mtok": 0.1,
      "output_usd_per_mtok": 0.4,
      "cached_input_usd_per_mtok": null,
      "note": "From the 'Token-based Pricing' table on Hyperstack's GPU pricing page.",
      "source_url": "https://www.hyperstack.cloud/gpu-pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gpt-oss-120b"
    },
    {
      "id": "hyperstack-llama-3.1-8b",
      "provider": "Hyperstack",
      "model": "Llama 3.1 8B",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.1-8B-Instruct",
      "input_usd_per_mtok": 0.2,
      "output_usd_per_mtok": 0.2,
      "cached_input_usd_per_mtok": null,
      "note": "From the 'Token-based Pricing' table on Hyperstack's GPU pricing page.",
      "source_url": "https://www.hyperstack.cloud/gpu-pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.1-8b"
    },
    {
      "id": "hyperstack-llama-3.3-70b",
      "provider": "Hyperstack",
      "model": "Llama 3.3 70B",
      "open_weights": true,
      "open_model_hint": "meta-llama/Llama-3.3-70B-Instruct",
      "input_usd_per_mtok": 0.8,
      "output_usd_per_mtok": 0.8,
      "cached_input_usd_per_mtok": null,
      "note": "From the 'Token-based Pricing' table on Hyperstack's GPU pricing page.",
      "source_url": "https://www.hyperstack.cloud/gpu-pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "llama-3.3-70b"
    },
    {
      "id": "crusoe-gemma-4-31b",
      "provider": "Crusoe",
      "model": "Gemma 4 31B-it",
      "open_weights": true,
      "open_model_hint": "google/gemma-4-31B-it",
      "input_usd_per_mtok": 0.14,
      "output_usd_per_mtok": 0.4,
      "cached_input_usd_per_mtok": 0.14,
      "note": "From the 'Serverless Inference pricing' table on Crusoe's cloud pricing page.",
      "source_url": "https://www.crusoe.ai/cloud/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gemma-4-31b"
    },
    {
      "id": "crusoe-glm-5.3",
      "provider": "Crusoe",
      "model": "GLM 5.3",
      "open_weights": true,
      "open_model_hint": "zai-org/GLM-5.3",
      "input_usd_per_mtok": 1.4,
      "output_usd_per_mtok": 4.4,
      "cached_input_usd_per_mtok": 0.26,
      "note": "From the 'Serverless Inference pricing' table on Crusoe's cloud pricing page.",
      "source_url": "https://www.crusoe.ai/cloud/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "glm-5.3"
    },
    {
      "id": "crusoe-gpt-oss-120b",
      "provider": "Crusoe",
      "model": "GPT-OSS 120B",
      "open_weights": true,
      "open_model_hint": "openai/gpt-oss-120b",
      "input_usd_per_mtok": 0.05,
      "output_usd_per_mtok": 0.2,
      "cached_input_usd_per_mtok": 0.05,
      "note": "From the 'Serverless Inference pricing' table on Crusoe's cloud pricing page.",
      "source_url": "https://www.crusoe.ai/cloud/pricing",
      "as_of": "2026-10-07",
      "preset_model_id": "gpt-oss-120b"
    }
  ]
}
