{"uid":"cap_PEdE-gVz5-5F0_d-usfKV","slug":"eval-engine-api-pay-per-call-ai-evaluation-6eb8de23","name":"Eval Engine API — Pay-per-call AI Evaluation","description":"Pay-per-call AI evaluation engine. Score LLM outputs, agent trajectories, and model responses against benchmark rubrics. $0.005 per eval via x402 USDC on Base. Free trial available.","url":"https://eval.zuluworksai.com/mcp","method":"POST","headers":{},"bodySchema":{"type":"object","required":["id","method","jsonrpc"],"properties":{"id":{"type":"string"},"method":{"type":"string","description":"MCP method (tools/call, tools/list, etc.)"},"params":{"type":"object"},"jsonrpc":{"type":"string"}}},"responseSchema":{"type":"object","properties":{"id":{"type":"string"},"error":{"type":"object"},"result":{"type":"object"},"jsonrpc":{"type":"string"}}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.005","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"down","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.005/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.005","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_vmD3Mlti6hTBcDsBYt1fX","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.005","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Scores LLM outputs, agent trajectories, and model responses against benchmark rubrics for automated AI quality evaluation","exampleAgentPrompt":"Score this LLM response against my accuracy rubric: the model said 'Paris is the capital of Germany' when asked about European capitals — I want to know how it performs on factual correctness and give me a numeric quality score.","exampleUseCases":[{"title":"Agent trajectory quality gate","prompt":"I need to evaluate whether my customer support agent handled this 10-turn conversation correctly — can you run it through an eval rubric that checks for helpfulness, accuracy, and appropriate escalation behavior and give me a score?"},{"title":"LLM output regression testing","prompt":"Before I deploy my new fine-tuned model, score these 5 sample responses against my benchmark rubric for tone, factual accuracy, and conciseness so I can see if it regressed from the previous version."},{"title":"Automated RAG answer grading","prompt":"Grade this RAG pipeline answer against the ground truth: the question was 'What is our refund policy?' and the model responded with a 3-sentence answer — evaluate it for faithfulness to the source document and completeness."}],"resultDescription":"Returns a JSON-RPC response object containing evaluation results including scores, rubric assessments, and quality ratings for the submitted LLM output or agent trajectory. The result field contains the structured evaluation with numeric scores and qualitative feedback per benchmark dimension.","failureModes":["Payment failure if insufficient USDC balance on Base — returns 402 Payment Required","Invalid JSON-RPC format causes parse error response with error object","Missing required fields (id, method, jsonrpc) returns validation error","Unknown MCP method returns method-not-found error","Malformed agent trajectory or LLM output may return low-confidence evaluation or error","Network timeout on complex trajectory evaluations"],"whenToPreferThis":"Choose this endpoint when you need pay-per-call, on-demand AI evaluation without committing to a subscription — ideal for CI/CD pipelines, spot-checking model outputs, or low-volume evaluation tasks. Best suited for teams that want to pay only for evals they run ($0.005 per call via USDC on Base) and need MCP-compatible tooling that integrates with agent frameworks. Prefer over batch eval platforms when you need real-time scoring in an agent loop.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T15:27:31.368Z","isFirstParty":false}