{"uid":"cap_dfy1ay9GviGGb3atabSDz","slug":"zeroreader-llama-3-2-1b-chat-completion-d59fbf92","name":"ZeroReader Llama 3.2 1B Chat Completion","description":"Llama 3.2 1B — Smallest Llama. Fastest, cheapest.","url":"https://api.zeroreader.com/v1/ai/llama-1b?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"stream":{"type":"boolean","default":false},"messages":{"type":"array","items":{"type":"object","required":["role","content"],"properties":{"role":{"enum":["system","user","assistant"],"type":"string"},"content":{"type":"string"}}}},"max_tokens":{"type":"integer","default":1024,"maximum":4096},"temperature":{"type":"number","default":0.7,"maximum":2,"minimum":0}}},"responseSchema":{"id":"chatcmpl-example","object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"I'm doing well!"},"finish_reason":"stop"}]},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.001","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":1,"rating":{"score":"1.00","successRate":"1.00","reviews":1,"stars":"5.0","state":"rated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.001/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.001","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.001","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_DZdOmpVMyD9nm5LwZPkVd","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.001","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Runs chat completions using Meta's Llama 3.2 1B model — the smallest and fastest Llama variant, optimized for low-latency, low-cost inference","exampleAgentPrompt":"Can you use the ZeroReader Llama 1B model to quickly answer this question for me: 'What are three common uses of the Python requests library?' — keep it under 200 tokens and use temperature 0.5 for a focused response.","exampleUseCases":null,"resultDescription":"Returns an OpenAI-compatible chat completion object containing the assistant's generated reply, a finish reason (e.g. 'stop'), a completion ID, and the full message object with role and content fields.","failureModes":["Temperature out of range (0-2) returns validation error","max_tokens exceeding 4096 is rejected","Malformed messages array missing required role or content fields causes error","Payment of 0.001 USDC not provided triggers 402 Payment Required","Empty messages array may return an error or empty completion"],"whenToPreferThis":"Choose this endpoint when you need the absolute fastest and cheapest LLM inference and your task is simple enough for a 1B parameter model — e.g. short Q&A, classification, simple summarization, or high-volume lightweight tasks where cost per call matters. Prefer larger sibling models (3B, 7B, 17B) when quality or reasoning depth is critical.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":1,"lastUsedAt":"2026-08-11T22:12:17.952Z","lastSuccessfullyRanAt":"2026-08-11T22:12:17.952Z","lastHealthCheckAt":"2026-10-02T00:31:16.023Z","isFirstParty":false,"canonicalSlug":"zeroreader-llama-3-2-1b-chat-completion-d59fbf92"}