{"uid":"cap_jVMJi0WRdzJ2P61TJUbrP","slug":"zeroreader-llama-3-3-70b-fast-chat-completion-0a092c4c","name":"ZeroReader Llama 3.3 70B (Fast) Chat Completion","description":"Llama 3.3 70B (Fast) — Flagship open-source LLM. Best quality on CF Workers AI.","url":"https://api.zeroreader.com/v1/ai/llama-70b?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"stream":{"type":"boolean","default":false},"messages":{"type":"array","items":{"type":"object","required":["role","content"],"properties":{"role":{"enum":["system","user","assistant"],"type":"string"},"content":{"type":"string"}}}},"max_tokens":{"type":"integer","default":1024,"maximum":4096},"temperature":{"type":"number","default":0.7,"maximum":2,"minimum":0}}},"responseSchema":{"id":"chatcmpl-example","object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"I'm doing well!"},"finish_reason":"stop"}]},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.008","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.008/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.008","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.008","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_Ntr7XH1Jr3ZfXFOx3lAak","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.008","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Runs inference against Meta's Llama 3.3 70B flagship open-source LLM via Cloudflare Workers AI, returning a chat completion response in OpenAI-compatible format","exampleAgentPrompt":"Send this conversation to Llama 3.3 70B and get a response — system prompt: 'You are a helpful assistant', user message: 'Explain the difference between supervised and unsupervised learning in plain English', with temperature 0.7 and up to 1024 tokens.","exampleUseCases":null,"resultDescription":"An OpenAI-compatible chat completion object containing a unique completion ID, the assistant's generated text in the message content field, a finish_reason (e.g. 'stop'), and a choice index. The response is a single JSON object (non-streaming by default).","failureModes":["Payment not provided or invalid x402 payment header — 402 Payment Required","Temperature out of range (>2 or <0) — validation error","max_tokens exceeds 4096 — validation error","Messages array missing required role or content fields — 400 Bad Request","Model overloaded or Cloudflare Workers AI backend unavailable — 503 Service Unavailable","Malformed JSON body — 400 Bad Request"],"whenToPreferThis":"Choose this endpoint when you need a high-quality, flagship open-source LLM (Llama 3.3 70B) with fast inference, pay-per-call USDC micropayment pricing (no subscription required), and OpenAI-compatible response format. Prefer this over smaller siblings (3B, 7B) when task complexity demands best-in-class open-source quality, and over reasoning-specialized siblings (DeepSeek R1) when you need general-purpose chat rather than chain-of-thought math/logic.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T01:16:30.540Z","isFirstParty":false,"canonicalSlug":"zeroreader-llama-3-3-70b-fast-chat-completion-0a092c4c"}