{"uid":"cap_49Pbx4VthBXRXGuxe2TEI","slug":"modelprices-xyz-cheapest-128k-context-llm-leaderboard-af1434ed","name":"modelprices.xyz Cheapest 128k-Context LLM Leaderboard","description":"Cheapest 128k-context LLM leaderboard: the 50 lowest-cost AI models with a 128,000-token context window or larger, ranked by inference cost per token across 70+ providers. Input, output and cache USD per 1M tokens with exact context window and capability flags joined in. The standard long-document tier, priced side-by-side. Refreshed hourly.","url":"https://modelprices.xyz/llm/cheapest/128k-context","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET"],"type":"string"},"queryParams":{"type":"object","properties":{}}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object"}}}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.01","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.01/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.01","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_x9aGgNz9k8SFeQ4WO-mys","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.01","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Returns the 50 lowest-cost AI models with at least a 128,000-token context window, ranked by inference cost per token across 70+ providers, with input/output/cache pricing in USD per 1M tokens.","exampleAgentPrompt":"Show me the 50 cheapest AI models that support at least a 128k context window, ranked by cost per token — I want to see input, output, and cache prices side by side so I can pick the most affordable one for processing long documents.","exampleUseCases":[{"title":"Cheapest model for long legal docs","prompt":"I need to process lengthy legal contracts — find me the cheapest AI models that can handle at least 128k tokens so I can pick the most cost-effective one for my pipeline."},{"title":"Budget LLM selection for RAG system","prompt":"I'm building a RAG system that needs a big context window. Pull up the cheapest 128k-context models ranked by price so I can decide which one fits my budget."},{"title":"Cost comparison before switching providers","prompt":"Before we switch our AI provider, can you check the current cheapest 128k-context LLMs across all providers and show me input and output pricing per million tokens?"}],"resultDescription":"A ranked list of up to 50 AI models, each with: model name, provider, input cost (USD/1M tokens), output cost (USD/1M tokens), cache cost (USD/1M tokens), exact context window size, and capability flags. Data is refreshed hourly.","failureModes":["Service temporarily unavailable (503) if the pricing data feed is being refreshed","Payment required error (402) if the $0.01 USDC per-call fee is not included","Empty or reduced result set if fewer than 50 models currently meet the 128k context threshold","Stale data risk within the hourly refresh window"],"whenToPreferThis":"Use this endpoint when you need to identify the most cost-efficient LLMs specifically for long-document or large-context workloads (128k+ tokens). Prefer this over the general cheapest-model endpoint when context window size is a hard constraint, and over provider-specific pricing tables when you want a cross-provider comparison in a single call. Ideal for automated model-selection pipelines, cost optimization workflows, and budget-constrained LLM deployments.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T00:46:11.335Z","isFirstParty":false}