{"uid":"cap_SPWEBVCCiA9-U9E4PmqY5","slug":"aayat-ai-fast-llm-chat-llama-3-2-3b-373558df","name":"Aayat AI Fast LLM Chat (Llama 3.2 3B)","description":"Pay-per-call LLM chat (Llama 3.2 3B): Fast, cheap model for classification, extraction, short answers and routing. No API key or account. POST JSON {\"prompt\": \"...\"} or OpenAI-style {\"messages\": [{\"role\": \"user\", \"content\": \"...\"}]}, optional \"system\", \"max_tokens\" (up to 1024), \"temperature\", \"json\": true. Up to about 15,000 characters of English in. Failed calls are not charged.","url":"https://aayatai.com/chat/fast?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"json":{"type":"boolean","default":false,"description":"Ask for JSON-only output."},"prompt":{"type":"string","maxLength":20000,"description":"A single user message (use this or messages)."},"system":{"type":"string","maxLength":8000,"description":"System instructions."},"messages":{"type":"array","items":{"type":"object","properties":{"role":{"enum":["system","user","assistant"],"type":"string"},"content":{"type":"string"}}},"maxItems":50,"description":"Conversation so far, OpenAI style: [{role, content}]."},"max_tokens":{"type":"integer","maximum":1024,"minimum":1,"description":"Most tokens to generate (default 512)."},"temperature":{"type":"number","default":0.3,"maximum":2,"minimum":0,"description":"Randomness, 0-2."}}},"responseSchema":{"type":"json","example":{"text":"x402 is an open protocol that lets clients pay for HTTP requests with stablecoins using the 402 status code.","model":"@cf/meta/llama-3.2-3b-instruct","usage":{"promptTokens":24,"completionTokens":26}}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.003","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.003/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.003","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.003","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_joRSEwngA1Gv7wupz5S7-","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.003","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Pay-per-call LLM inference using Llama 3.2 3B for fast, cheap text classification, extraction, short-answer generation, and routing tasks — no API key required.","exampleAgentPrompt":"Classify this customer message as 'complaint', 'question', or 'compliment' and return only valid JSON with a 'label' field: 'My order arrived two days late and the packaging was damaged.'","exampleUseCases":[{"title":"Intent routing in an agent pipeline","prompt":"Figure out whether this user message is asking to book a flight, check a balance, or do something else entirely — just return the intent label as JSON: 'Can you show me my account statement from last month?'"},{"title":"Structured data extraction from text","prompt":"Pull out the product name, price, and quantity from this order confirmation snippet and give me the result as JSON: 'You ordered 3x Wireless Earbuds Pro at $49.99 each, total $149.97.'"},{"title":"Quick factual short-answer generation","prompt":"In one or two sentences, explain what the x402 payment protocol is — keep it concise, plain English, no more than 100 tokens."}],"resultDescription":"Returns a JSON object containing the generated text under 'text', the model identifier under 'model' (e.g. '@cf/meta/llama-3.2-3b-instruct'), and a 'usage' object with 'promptTokens' and 'completionTokens' counts. When json mode is enabled, the 'text' field contains valid JSON output.","failureModes":["Prompt or messages exceeding ~15,000 characters may be rejected or truncated","max_tokens above 1024 will be rejected by schema validation","Malformed request body (missing both prompt and messages) returns an error and is not charged","Ambiguous or very long prompts may produce truncated or low-quality outputs due to the small 3B model size","Network timeouts on the caller side; failed calls are not billed"],"whenToPreferThis":"Choose this endpoint when you need fast, cheap LLM inference for simple tasks like classification, short extraction, or routing and want to avoid API key provisioning or account setup. It is ideal for high-volume, low-complexity agentic subtasks where cost per call matters and a 3B-parameter model is sufficient. Prefer larger hosted LLMs (GPT-4, Claude) when the task requires deep reasoning, long context, or high accuracy on complex language understanding.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T00:44:38.214Z","isFirstParty":false,"canonicalSlug":"aayat-ai-fast-llm-chat-llama-3-2-3b-373558df"}