{"uid":"cap_d_4R_lMXHPVlOQNDVekyY","slug":"pqs-onchainintel-net-a370506f","name":"PQS Cross-Model Scoring: Claude vs GPT-4o Comparison","description":"PQS cross-model scoring - same prompt through Claude and GPT-4o, judged by a third model","url":"https://pqs.onchainintel.net/api/score/compare","method":"GET","headers":{},"bodySchema":{"type":"object","properties":{"prompt":{"type":"string","maxLength":10000,"minLength":1,"description":"Prompt to score"},"vertical":{"enum":["software","content","business","education","science","crypto","general","research"],"type":"string","default":"general","description":"Domain: software/content/business/education/science/crypto/general/research"}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"1.25","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$1.25/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"1.25","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"1.25","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_km-g7604dY7ods6ugwJnr","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"1.25","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Runs the same prompt through both Claude and GPT-4o and uses a third model as judge to score and compare the outputs","exampleAgentPrompt":"Run my prompt 'Explain how garbage collection works in Python' through both Claude and GPT-4o and use a third model to judge which one gives the better answer — the domain is software.","exampleUseCases":null,"resultDescription":"Returns a scored comparison of Claude and GPT-4o outputs for the same prompt, with a third model acting as judge to evaluate quality differences, scores, and which model performed better in the given vertical.","failureModes":["Missing or empty prompt returns validation error","Invalid vertical enum value causes bad request","One upstream model (Claude or GPT-4o) unavailable causes partial or failed comparison","Prompt too long for one or both models causes truncation or error","Payment not included or invalid x402 header causes 402 Payment Required"],"whenToPreferThis":"Use this endpoint when you need an objective, third-model-judged comparison of Claude vs GPT-4o on a specific prompt — ideal for prompt engineering research, selecting the best model for a domain, or validating which LLM to use before committing to a production integration. Prefer this over single-model scoring endpoints when the goal is model selection rather than prompt quality alone.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-15T06:36:41.119Z","isFirstParty":false}