{"uid":"cap_Y58kF7BFwMyM9CM_vT5Lr","slug":"signalharness-llm-eval-regression-report-c805404d","name":"SignalHarness LLM Eval Regression Report","description":"Explore 330 pay-per-call x402 API services and 27 agent-native digital products, with Base USDC pricing, secure Polar checkout, and free discovery.","url":"https://signalharness.ai/api/agent/services/llm_eval_regression_report/invoke","method":"POST","headers":{},"bodySchema":{"type":"object","properties":{"request_json":{"type":"string","maxLength":65536,"minLength":2}}},"responseSchema":{"type":"json","example":{"replay":false,"result":{"warnings":["Verify the caller-supplied data before relying on this result."],"service_id":"llm_eval_regression_report","analysis_json":"{\"example\":\"schema-valid caller-supplied data\"}","evidence_scope":"caller_supplied_data"},"status":"succeeded","receipt":{"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","usage":[],"status":"succeeded","network":"eip155:8453","artifacts":[],"endedAtMs":0,"latencyMs":0,"paymentId":"example-payment","receiptId":"example-receipt","requestId":"example-request","serviceId":"llm_eval_regression_report","executionId":"example-execution","startedAtMs":0,"amountAtomic":"25000","resultSha256":"d376eb6705d37d8522d0688051a70b42c49c43a5713e4a3864904938432c28b3","serviceVersion":"1.0.0","settlementReference":"0x0000000000000000000000000000000000000000000000000000000000000000"},"artifacts":[],"requestId":"example-request","executionId":"example-execution"}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.025","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.025/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.025","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.025","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_nuaqEYjiIbrxmttG63A10","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.025","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Generates a regression analysis report for LLM evaluation results, detecting quality regressions across model versions or prompts.","exampleAgentPrompt":"I ran evaluations on two versions of my language model and want to know if the new version regressed — can you submit this eval JSON to SignalHarness and get back a regression report showing where quality dropped?","exampleUseCases":[{"title":"Pre-deployment model quality gate","prompt":"Before I ship the new version of my fine-tuned model, run this eval results JSON through the regression checker and tell me if anything got meaningfully worse compared to baseline."},{"title":"Prompt engineering regression check","prompt":"I changed several system prompts in my pipeline and here are the before-and-after evaluation results as JSON — can you analyze them for regressions and summarize which prompts caused quality drops?"},{"title":"Continuous eval monitoring report","prompt":"Here's the latest nightly LLM eval output JSON from our CI pipeline — run a regression report on it and flag any test cases that degraded since last week's run."}],"resultDescription":"Returns a JSON object containing an analysis_json field with the structured regression findings, a warnings array (e.g. reminders to verify caller-supplied data), an evidence_scope label indicating the data provenance, and a receipt object with payment and execution metadata including service version, latency, and a SHA-256 result hash for integrity verification.","failureModes":["Malformed or too-short request_json (below 2 chars) returns a validation error","Payload exceeding 65536 characters is rejected","Caller-supplied data that does not conform to expected eval schema may produce unreliable analysis with warnings","Payment failure via x402 protocol results in a 402 response before execution","Network or service timeout returns an error status in the receipt"],"whenToPreferThis":"Choose this endpoint when you need a structured, pay-per-call regression analysis of LLM evaluation data without standing up your own eval infrastructure. It is particularly well-suited for CI/CD pipelines, pre-deployment quality gates, or ad hoc regression checks on prompt changes, and is priced at $0.025 USDC per call making it cost-effective for frequent automated use.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T18:48:10.984Z","isFirstParty":false}