{"uid":"cap_RitvtPdAANnOtteSeQrov","slug":"tenjin-ml-evaluation-claims-expire-paper-trail-ff5ae6e1","name":"Tenjin: ML Evaluation Claims Expire (Paper Trail)","description":"Two July papers argue that benchmark scores need scope and expiration metadata; WorkSurface-Bench, a synthetic-user benchmark, a coding-agent user study, and a long-context aggregation method show what gets lost when a score is treated as portable evidence.","url":"https://tenjin.blog/api/read/arxiv-ml/paper-trail-evaluation-claims-expire","method":"GET","headers":{},"bodySchema":{"type":"object","$schema":"https://json-schema.org/draft/2020-12/schema","required":["input"],"properties":{"input":{"type":"object","required":["type","method"],"properties":{"type":{"type":"string","const":"http"},"method":{"enum":["GET"],"type":"string"},"pathParams":{"type":"object","required":["handle","slug"],"properties":{"slug":{"type":"string","description":"The article's URL slug, unique per creator. The reserved slug `latest` resolves to the creator's newest published piece; its stable scheduled-read form is the wallet-address URL /api/read/<0x-address>/latest (a handle `latest` is not payable)."},"handle":{"type":"string","description":"The creator's handle, or their wallet address. The address form is REQUIRED for a durable `latest` alias (a handle `latest` is not payable), and is the only form for an unclaimed creator."}}},"queryParams":{"type":"object","required":[],"properties":{},"additionalProperties":false}},"additionalProperties":false},"output":{"type":"object","required":["type"],"properties":{"type":{"type":"string"},"example":{"type":"object"}}}}},"responseSchema":null,"example":null,"exampleRequest":null,"tags":["x402"],"displayCostAmount":"0.1","displayCostAsset":"USDC","priceDynamic":false,"priceHint":null,"priceStatus":"priced","priceSource":"probe","requiresHandshake":false,"reviewCount":0,"rating":{"score":"0.00","successRate":"0.00","reviews":0,"stars":null,"state":"unrated"},"availabilityStatus":"unknown","priceObserved":null,"sessionDeposit":null,"pricing":{"kind":"static","summary":"$0.1/call","primary":{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.1","per":"call","confidence":"exact"},"accepted":[{"kind":"static","protocol":"x402","network":"base","amountUsd":"0.1","per":"call","confidence":"exact"}]},"paymentMethods":[{"uid":"pm_6Hgz1laJHh1eiw3azCUB5","protocol":"x402","methodType":"crypto","chain":"base","mode":"charge","costAmount":"0.1","costPer":"request","priority":0,"asset":"0x833589fCD6eDb6E08f4c7C32D4f71b54bdA02913","unit":"request","depositMicros":null,"planRef":null}],"brandName":null,"brandSlug":null,"brandBaseUrl":null,"brandDocsUrl":null,"whatItDoes":"Returns a curated analysis of two July ML papers arguing that benchmark scores need scope and expiration metadata, with coverage of WorkSurface-Bench, a synthetic-user benchmark, a coding-agent user study, and a long-context aggregation method.","exampleAgentPrompt":"Fetch the Tenjin article on why ML evaluation claims expire — the one covering WorkSurface-Bench, synthetic-user benchmarks, and long-context aggregation from the arxiv-ml paper-trail series.","exampleUseCases":[{"title":"Researching ML benchmark reliability","prompt":"Pull up the Tenjin paper trail article about how benchmark scores need expiration metadata — I want to understand what WorkSurface-Bench shows about treating scores as portable evidence."},{"title":"Staying current on AI evaluation methods","prompt":"Get me the Tenjin article covering those two July ML papers arguing benchmark scores should have scope and expiration dates attached — the one from the arxiv-ml feed."},{"title":"Understanding coding-agent evaluation gaps","prompt":"Fetch the Tenjin piece that includes the coding-agent user study and long-context aggregation method — I'm trying to understand what gets lost when benchmark scores are reused out of context."}],"resultDescription":"A structured article object containing the full content of the Tenjin piece on ML evaluation claims expiring, including analysis of two July papers, discussion of WorkSurface-Bench, a synthetic-user benchmark, a coding-agent user study, and a long-context aggregation method, along with metadata about the publication.","failureModes":["Article not found if slug or handle is incorrect — returns 404 or error object","Payment failure if 0.1 USDC x402 payment is not properly attached — returns 402 Payment Required","Handle resolves to wrong creator if mistyped — returns unexpected content","Rate limiting or access errors if payment protocol is misconfigured"],"whenToPreferThis":"Use this endpoint when you specifically need the content of this Tenjin article about ML benchmark expiration and scope metadata. Prefer this over general web search when you need structured, paywall-gated article content from Tenjin's arxiv-ml paper trail series in a machine-readable format.","instructions":null,"reviewSummary":null,"reviewSummaryHighlights":null,"reviewSummaryConcerns":null,"reviewSummaryGeneratedAt":null,"activationCount":0,"lastUsedAt":null,"lastSuccessfullyRanAt":null,"lastHealthCheckAt":"2026-09-13T13:02:26.193Z","isFirstParty":false}