{"uid":"cap_GcgZfH-Df_GepKKLphbug","slug":"zeroreader-llama-3-2-11b-vision-028971c4","name":"ZeroReader Llama 3.2 11B Vision","description":"Llama 3.2 11B Vision — Vision + text model. Can understand images.","url":"https://api.zeroreader.com/v1/ai/llama-vision?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"stream":{"type":"boolean","default":false},"messages":{"type":"array","items":{"type":"object","required":["role","content"],"properties":{"role":{"enum":["system","user","assistant"],"type":"string"},"content":{"type":"string"}}}},"max_tokens":{"type":"integer","default":1024,"maximum":4096},"temperature":{"type":"number","default":0.7,"maximum":2,"minimum":0}}},"responseSchema":{"id":"chatcmpl-example","object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"I'm doing well!"},"finish_reason":"stop"}]},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.005","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.005/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_Q3dwVf1QR4MYsoAjL35Kc","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.005","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Runs multimodal chat completions using Meta's Llama 3.2 11B Vision model, capable of understanding both images and text","exampleAgentPrompt":"Look at this image of a product label and tell me all the ingredients listed — use the vision model with temperature 0.3 and up to 512 tokens in your response.","exampleUseCases":null,"resultDescription":"Returns a chat completion object in OpenAI-compatible format, including the assistant's message content, finish reason (e.g. 'stop'), and a completion ID. The assistant's response will reflect analysis of any image or text provided in the messages array.","failureModes":["Invalid or missing messages array returns a 400 validation error","Temperature outside 0–2 range causes a 400 error","max_tokens exceeding 4096 returns a 400 error","Image content that cannot be parsed or is too large may result in a 422 or model error","Payment failure (x402) returns a 402 Payment Required before processing","Network timeout if the model takes too long to generate a long response"],"whenToPreferThis":"Choose this endpoint when you need a vision-capable model that can process both images and text in a single request. Prefer this over text-only Llama variants (3B, etc.) when the input contains visual content like photos, screenshots, charts, or diagrams. Prefer over larger models when cost and speed matter and the task doesn't require deep reasoning or 100B+ parameter capacity.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T01:28:46.037Z","isFirstParty":false,"canonicalSlug":"zeroreader-llama-3-2-11b-vision-028971c4"}