{"uid":"cap_bBPSwl_H868XVOOnggFqA","slug":"xona-agent-llm-inference-nvidia-nim-openrouter-fallback-48c49d35","name":"Xona Agent LLM Inference (NVIDIA NIM / OpenRouter Fallback)","description":"Frontier-model inference with automatic provider fallback (NVIDIA NIM → OpenRouter) for reliability. OpenAI-compatible chat completions; select a model via the body `model` field. Usage-based pricing (input + output tokens at provider cost + 10%); the x402 per-call charge is quoted up front as an upper bound, so set max_tokens to lower it. Also available drop-in at POST /v1/chat/completions (API key + credits, billed exactly per token).","url":"https://api.xona-agent.com/llm/nim?utm_source=zero.xyz","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"model":{"type":"string","default":"deepseek-v4-pro","description":"Model id (usage-based price: input + output tokens at provider cost + 10%). One of: kimi-k2.6, glm-5.1, nemotron-3-ultra-550b, deepseek-v4-pro"},"prompt":{"type":"string","description":"The user prompt (or use messages)"},"messages":{"type":"array","items":{"type":"string"},"description":"Optional OpenAI-style messages array (overrides prompt/system_prompt)"},"max_tokens":{"type":"number","description":"Cap on output tokens (max 8192). Lowers the up-front per-call charge — unset quotes the full cap as an upper bound."},"temperature":{"type":"number","description":"Optional sampling temperature"},"system_prompt":{"type":"string","description":"Optional system prompt"}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.007844","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.015/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.015","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.015","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_j19mN9R7-mWfsS-_XoAOV","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.015","costPer":"request","priority":0,"asset":null,"unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"OpenAI-compatible chat completions with automatic provider fallback (NVIDIA NIM → OpenRouter), supporting frontier models like DeepSeek, Kimi, GLM, and Nemotron via pay-per-call x402 pricing.","exampleAgentPrompt":"Using the deepseek-v4-pro model, answer this question with up to 512 output tokens: 'Explain the difference between gradient descent and stochastic gradient descent in simple terms.'","exampleUseCases":null,"resultDescription":"An OpenAI-compatible chat completion response containing the assistant's generated text, along with token usage details (input + output tokens). The response follows the standard ChatCompletion object format.","failureModes":["Model not available — if NVIDIA NIM is unavailable and OpenRouter fallback also fails, returns an error","max_tokens exceeded — request rejected if max_tokens > 8192","Invalid model ID — returns error if model field does not match one of the four supported model IDs","Payment failure — x402 charge not resolved, request blocked","Rate limiting — too many concurrent requests may be throttled"],"whenToPreferThis":"Prefer this endpoint when you need frontier model inference (DeepSeek-v4-pro, Nemotron-3-ultra-550b, Kimi-k2.6, GLM-5.1) with high reliability via automatic provider fallback, and want pay-per-call x402 micropayment pricing instead of a subscription. Ideal for agents that need sporadic, usage-based LLM calls without managing separate provider API keys.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-10-02T04:57:04.284Z","isFirstParty":false,"canonicalSlug":"xona-agent-llm-inference-nvidia-nim-openrouter-fallback-48c49d35"}