{"uid":"cap_OqFAr17Tv4SOrjEsPutjU","slug":"llm-chat-completion-via-unykorn-genesis402-local-rtx-5090-hosted-395dff21","name":"LLM Chat Completion via UnyKorn / Genesis402 (Local RTX 5090 + Hosted Fallback)","description":"LLM chat completion (local RTX 5090 models first, hosted allowlist second) — Genesis402 / UnyKorn Operator Network","url":"https://twin.unykorn.org/llm?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"params":{"type":"object","properties":{"model":{"type":"string"},"prompt":{"type":"string","maxLength":24000},"messages":{"type":"array","items":{"type":"object","required":["role","content"],"properties":{"role":{"enum":["system","user","assistant"],"type":"string"},"content":{"type":"string"}}}},"max_tokens":{"type":"integer","maximum":1024,"minimum":1},"temperature":{"type":"number","maximum":2,"minimum":0}}}}},"responseSchema":{"type":"json","example":{"ok":true,"type":"llm","model":"qwen2.5:7b","usage":{"prompt_tokens":42,"completion_tokens":88},"output":"<assistant text>","receipt":{"tx_hash":"0x<64hex>","amount_usd":0.002,"receipt_id":"g402-<16hex>"},"provider":"ollama-local","finish_reason":"stop"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.002","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.002/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.002","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_QMI5Nhsd7ujDMtyfgBsZh","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.002","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Runs LLM chat completions preferring local RTX 5090 GPU models, falling back to an allowlisted hosted provider, billed per-call via x402 micropayments.","exampleAgentPrompt":"Using the local RTX 5090 LLM endpoint, send these messages to qwen2.5:7b with a temperature of 0.7 and max 512 tokens: system says 'You are a helpful assistant', user asks 'Explain the difference between proof-of-work and proof-of-stake in two sentences.'","exampleUseCases":[{"title":"Autonomous agent reasoning step","prompt":"Send this to the local GPU model — system: 'You are a concise reasoning engine', user: 'Given these three options, which is the most cost-efficient: AWS Lambda, a VPS, or a local server? Reply in under 100 words.' Use qwen2.5:7b, temperature 0.3, max 200 tokens."},{"title":"Customer-facing chatbot reply generation","prompt":"I need a customer support reply generated — use the local LLM with temperature 0.5 and up to 300 tokens. System message: 'You are a polite e-commerce support agent.' User message: 'My order hasn't arrived after 10 days, what should I do?'"},{"title":"Code snippet generation on-device","prompt":"Ask the local RTX 5090 model to write me a Python function that parses a JSON array and returns only the items where the key 'active' is true. Use qwen2.5:7b, max 400 tokens, temperature 0.2."}],"resultDescription":"Returns a JSON object with the assistant's generated text, the model that served the request, prompt and completion token counts, finish reason, the provider label (e.g. 'ollama-local'), and a receipt object containing the transaction hash, USD amount charged, and a receipt ID for audit purposes.","failureModes":["Model not available locally and not on hosted allowlist — returns error or falls back unexpectedly","Prompt exceeds 24,000 character limit — request rejected","max_tokens out of range (must be 1–1024) — validation error","x402 payment failure or insufficient balance — payment not settled, request not processed","Temperature out of range (0–2) — validation error","Timeout if local GPU is busy under load"],"whenToPreferThis":"Choose this endpoint when you want pay-per-call LLM inference without a subscription, prefer local GPU execution for privacy or latency, want verifiable on-chain payment receipts via x402, or are building an agent that needs lightweight chat completions billed in USDC micropayments. Prefer over hosted OpenAI-style APIs when cost-per-call transparency and decentralized billing matter.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T18:59:55.244Z","isFirstParty":false,"canonicalSlug":"llm-chat-completion-via-unykorn-genesis402-local-rtx-5090-hosted-395dff21"}