{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:TIWN2SGFVWMJJHXZKCGGZOGRYL","short_pith_number":"pith:TIWN2SGF","schema_version":"1.0","canonical_sha256":"9a2cdd48c5ad98949ef9508c6cb8d1c2e4fc3c5d52b7461b5b158f66ac071cb0","source":{"kind":"arxiv","id":"2607.17409","version":1},"attestation_state":"computed","paper":{"title":"Efficient Sequential Evaluation of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ME"],"primary_cat":"stat.ML","authors_text":"Chia-Yu Hsu, Shubhanshu Shekhar","submitted_at":"2026-07-19T21:01:34Z","abstract_excerpt":"We study the problem of sequentially evaluating a new large language model (LLM) on a fixed question set using historical performance data from prior LLMs. Our goal is to construct a confidence sequence (CS) for the model's capability on this question set and to design active querying rules that shrink the CS width as quickly as possible. For CS construction, we invert a family of test supermartingales and focus on two representative approaches: a reverse information projection (RIPr)-based approach and a testing-by-betting-based approach. We first study these approaches under an oracle settin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.17409","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2026-07-19T21:01:34Z","cross_cats_sorted":["cs.LG","stat.ME"],"title_canon_sha256":"19e630e80abf4d474964e9d3b728ee1dce3206199374c5b63f3d0e8a8d858fc3","abstract_canon_sha256":"25f00fbed744e1b9e2055b02bf121c238e2d80e81b8914b9d1e13e7235a637e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T01:21:31.530424Z","signature_b64":"KMWsbml1uTDxrGHrBedVHy9i1WDWp2LQp9dvYnKcdGBmY7i7rLVE+WB3Fsk3FTLJX0fTtkTh3sub9k2R3iLXDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a2cdd48c5ad98949ef9508c6cb8d1c2e4fc3c5d52b7461b5b158f66ac071cb0","last_reissued_at":"2026-07-21T01:21:31.529537Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T01:21:31.529537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Sequential Evaluation of Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","stat.ME"],"primary_cat":"stat.ML","authors_text":"Chia-Yu Hsu, Shubhanshu Shekhar","submitted_at":"2026-07-19T21:01:34Z","abstract_excerpt":"We study the problem of sequentially evaluating a new large language model (LLM) on a fixed question set using historical performance data from prior LLMs. Our goal is to construct a confidence sequence (CS) for the model's capability on this question set and to design active querying rules that shrink the CS width as quickly as possible. For CS construction, we invert a family of test supermartingales and focus on two representative approaches: a reverse information projection (RIPr)-based approach and a testing-by-betting-based approach. We first study these approaches under an oracle settin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.17409","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.17409/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.17409","created_at":"2026-07-21T01:21:31.529990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.17409v1","created_at":"2026-07-21T01:21:31.529990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.17409","created_at":"2026-07-21T01:21:31.529990+00:00"},{"alias_kind":"pith_short_12","alias_value":"TIWN2SGFVWMJ","created_at":"2026-07-21T01:21:31.529990+00:00"},{"alias_kind":"pith_short_16","alias_value":"TIWN2SGFVWMJJHXZ","created_at":"2026-07-21T01:21:31.529990+00:00"},{"alias_kind":"pith_short_8","alias_value":"TIWN2SGF","created_at":"2026-07-21T01:21:31.529990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL","json":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL.json","graph_json":"https://pith.science/api/pith-number/TIWN2SGFVWMJJHXZKCGGZOGRYL/graph.json","events_json":"https://pith.science/api/pith-number/TIWN2SGFVWMJJHXZKCGGZOGRYL/events.json","paper":"https://pith.science/paper/TIWN2SGF"},"agent_actions":{"view_html":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL","download_json":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL.json","view_paper":"https://pith.science/paper/TIWN2SGF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.17409&json=true","fetch_graph":"https://pith.science/api/pith-number/TIWN2SGFVWMJJHXZKCGGZOGRYL/graph.json","fetch_events":"https://pith.science/api/pith-number/TIWN2SGFVWMJJHXZKCGGZOGRYL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL/action/storage_attestation","attest_author":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL/action/author_attestation","sign_citation":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL/action/citation_signature","submit_replication":"https://pith.science/pith/TIWN2SGFVWMJJHXZKCGGZOGRYL/action/replication_record"}},"created_at":"2026-07-21T01:21:31.529990+00:00","updated_at":"2026-07-21T01:21:31.529990+00:00"}