{"uid":"cap__ymt6Y6X6zB-T_hSMMnPG","slug":"pqs-compare-head-to-head-prompt-scoring-across-models-83e2f4b1","name":"PQS Compare - Head-to-Head Prompt Scoring Across Models","description":"A Quality Gate For Prompts. Before they break production. Score any AI prompt against 8 dimensions, see what&#x27;s weak, and get the fixed version in seconds.","url":"https://promptqualityscore.com/api/score/compare","method":"GET","headers":{},"bodySchema":{"type":"object","properties":{"prompt":{"type":"string","maxLength":10000,"minLength":1,"description":"Prompt to score"},"vertical":{"enum":["software","content","business","education","science","crypto","general","research"],"type":"string","default":"general","description":"Domain: software/content/business/education/science/crypto/general/research"}}},"responseSchema":{"type":"json","example":{"winner":"claude","results":{"gpt4o":{"model":"gpt-4o","output":"PQS measures prompt quality across 8 dimensions before the prompt reaches the model - a pre-flight gate for AI-agent buyers.","scores":{"total":32,"relevancy":9,"completeness":8,"faithfulness":8,"reasoning_depth":7}},"claude":{"model":"claude-sonnet-4-6","output":"PQS scores any LLM prompt on 8 dimensions pre-inference so x402 API buyers can reject low-quality inputs before paying.","scores":{"total":35,"relevancy":9,"completeness":9,"faithfulness":9,"reasoning_depth":8}}},"verdict":"Claude's response names the exact buyer action (reject) and ties to x402, giving the technical buyer a concrete decision rule.","vertical":"general","powered_by":"PQS - promptqualityscore.com","pqs_version":"2.0","original_prompt":"Explain PQS scoring in one sentence to an AI agent operator.","optimized_prompt":"You are an AI-agent product manager. In one sentence (max 25 words), explain PQS scoring to a technical buyer evaluating x402 APIs."}},"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"1.25","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$1.25/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"1.25","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"1.25","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_7473IZw59shtTNG8Bv8XD","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"1.25","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Scores a prompt against 8 quality dimensions, runs it through multiple LLMs (GPT-4o and Claude), compares outputs with scores, and returns an optimized prompt with a winner verdict.","exampleAgentPrompt":"Run a PQS compare on this prompt against the software vertical and tell me which model wins and what the optimized version looks like: 'You are a senior engineer. Review the following code diff and list any bugs, performance issues, and security vulnerabilities in order of severity.'","exampleUseCases":null,"resultDescription":"Returns a JSON object with: the winning model name, per-model outputs and dimension scores (relevancy, completeness, faithfulness, reasoning_depth, total), a natural-language verdict explaining why one model outperformed the other, the original prompt, and an AI-optimized rewrite of the prompt. Also includes PQS version and the vertical used.","failureModes":["Empty or missing prompt field returns validation error","Prompt exceeds 10,000 character limit","Invalid vertical enum value causes rejection","Payment not included or insufficient (x402 payment required at $1.25 USDC)","LLM provider timeout causes incomplete comparison results","Network error from upstream model providers"],"whenToPreferThis":"Use this endpoint when you need to compare how multiple LLMs handle the same prompt side-by-side, get dimension-level quality scores, and receive an auto-optimized prompt rewrite — all in one call. Prefer this over the basic scoring endpoint when model selection matters or when you want a concrete recommendation on which model to use for a given prompt. Ideal for prompt QA pipelines before deploying to production, or when evaluating a new prompt for an AI agent.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T12:36:02.892Z","isFirstParty":false}