{"uid":"cap_frJ_UC6Xe-f7RmqpCWRWU","slug":"zeroreader-llama-3-2-3b-chat-completion-a12ff00c","name":"ZeroReader Llama 3.2 3B Chat Completion","description":"Llama 3.2 3B — Good balance of speed and quality for simple tasks.","url":"https://api.zeroreader.com/v1/ai/llama-3b?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"stream":{"type":"boolean","default":false},"messages":{"type":"array","items":{"type":"object","required":["role","content"],"properties":{"role":{"enum":["system","user","assistant"],"type":"string"},"content":{"type":"string"}}}},"max_tokens":{"type":"integer","default":1024,"maximum":4096},"temperature":{"type":"number","default":0.7,"maximum":2,"minimum":0}}},"responseSchema":{"id":"chatcmpl-example","object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"I'm doing well!"},"finish_reason":"stop"}]},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.002","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.002/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_MOYar4UCpohuEkLwKraqT","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.002","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Runs chat completions using Meta's Llama 3.2 3B model, offering a fast and cost-effective balance of speed and quality for straightforward text generation tasks.","exampleAgentPrompt":"Send this conversation to the Llama 3.2 3B model with a system message saying 'You are a helpful assistant' and a user message asking 'What are three tips for writing clean Python code?' — keep temperature at 0.7 and max tokens at 512.","exampleUseCases":null,"resultDescription":"Returns an OpenAI-compatible chat completion object with a choices array containing the assistant's generated message, a finish_reason ('stop' or 'length'), and a completion ID. The assistant content is a plain text string.","failureModes":["Invalid or missing 'messages' array returns a 400 validation error","Temperature outside [0, 2] range causes a schema validation rejection","max_tokens exceeding 4096 is rejected with a parameter error","Payment failure or insufficient USDC balance results in a 402 Payment Required response","Model unavailability or overload may return a 503 or timeout","Malformed role values (not 'system', 'user', or 'assistant') cause a 400 error"],"whenToPreferThis":"Choose this endpoint when you need fast, low-cost chat completions for simple or short-form tasks where a 3B parameter model is sufficient — e.g., classification, summarization, Q&A, or lightweight generation. Prefer it over the larger sibling models (17B, 32B, 120B) when latency and cost matter more than peak capability. Avoid it for complex reasoning, math, or code tasks where DeepSeek R1 32B or GPT-OSS 120B would be more appropriate.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T01:18:30.535Z","isFirstParty":false,"canonicalSlug":"zeroreader-llama-3-2-3b-chat-completion-a12ff00c"}