{"uid":"cap_XaxccmRWsqXpGcVJH4zk9","slug":"dcl-trust-oracle-jailbreak-prompt-injection-detector-7f5d34b8","name":"DCL Trust Oracle — Jailbreak & Prompt Injection Detector","description":"Detect jailbreaks, prompt injection, and instruction conflicts before the agent follows them.","url":"https://bazaar.fronesislabs.com/evaluate/jailbreak","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET","HEAD","DELETE"],"type":"string"},"queryParams":{"type":"object","properties":{}}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object","required":["verdict","confidence","reason","tx_hash","chain_index"],"properties":{"reason":{"type":"string"},"tx_hash":{"type":"string"},"verdict":{"type":"string"},"timestamp":{"type":"number"},"confidence":{"type":"number"},"drift_mode":{"type":"string"},"input_hash":{"type":"string"},"chain_index":{"type":"integer"},"drift_score":{"type":"number"},"policy_version":{"type":"string"}}}}}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.02","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.02/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.02","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_opLs-xfv6eswpzTSmMu9i","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.02","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Evaluates agent inputs for jailbreaks, prompt injection attempts, and instruction conflicts before execution","exampleAgentPrompt":"Before I process this user message, run it through the DCL Trust Oracle jailbreak evaluator to check if it contains prompt injection or instruction conflicts — give me the verdict, confidence score, and reason.","exampleUseCases":[{"title":"Guarding autonomous agent from hijacking","prompt":"Before my agent acts on this incoming user instruction, screen it for jailbreak attempts or prompt injection and tell me the verdict and confidence level so I can decide whether to proceed."},{"title":"Financial agent pre-action safety check","prompt":"This user just sent an instruction to my payment agent — can you run it through the Trust Oracle jailbreak detector to check for adversarial manipulation before we execute any transactions?"},{"title":"Detecting instruction conflicts in multi-agent pipeline","prompt":"I have a message coming into my agent pipeline that looks suspicious — check it for prompt injection, instruction conflicts, and tell me the drift score and reason so I can block it if needed."}],"resultDescription":"Returns a verdict (e.g. 'safe', 'jailbreak', 'injection'), a confidence score (0-1), a human-readable reason, a drift score and drift mode indicating behavioral deviation, the policy version used, a hash of the input, a blockchain transaction hash for auditability, a chain index, and a timestamp.","failureModes":["Missing required 'input' parameter returns validation error","Unsupported HTTP method (only GET/HEAD/DELETE accepted) returns 405","Payment not provided or insufficient USDC results in 402 Payment Required","Malformed query parameters may return 400 Bad Request","Service unavailability returns 503 with no verdict"],"whenToPreferThis":"Choose this endpoint when you need lightweight, fast pre-execution screening of agent inputs for adversarial manipulation — specifically jailbreaks, prompt injection, and instruction conflicts — before an agent acts. It is most valuable in agentic pipelines where untrusted user input feeds into autonomous decision-making, especially for irreversible or sensitive actions. Prefer this over general content moderation when your threat model centers on adversarial prompt-level attacks rather than harmful content categories.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-14T00:43:31.010Z","isFirstParty":false}