{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TIBGYRW2O4Z5752YKOAGOMJE4O","short_pith_number":"pith:TIBGYRW2","schema_version":"1.0","canonical_sha256":"9a026c46da7733dff7585380673124e393edd6bbba36e06ade8bb87014fef2ba","source":{"kind":"arxiv","id":"2506.07949","version":1},"attestation_state":"computed","paper":{"title":"Cost-Optimal Active AI Model Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Fisch, Alekh Agarwal, Anastasios N. Angelopoulos, Jacob Eisenstein, Jonathan Berant","submitted_at":"2025-06-09T17:14:41Z","abstract_excerpt":"The development lifecycle of generative AI systems requires continual evaluation, data acquisition, and annotation, which is costly in both resources and time. In practice, rapid iteration often makes it necessary to rely on synthetic annotation data because of the low cost, despite the potential for substantial bias. In this paper, we develop novel, cost-aware methods for actively balancing the use of a cheap, but often inaccurate, weak rater -- such as a model-based autorater that is designed to automatically assess the quality of generated content -- with a more expensive, but also more acc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07949","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-09T17:14:41Z","cross_cats_sorted":[],"title_canon_sha256":"ee713f322985cb177b073e149cfdaa7b6b229f2f596bb7ef43b6a7fc9c94cb72","abstract_canon_sha256":"1a87397f70e4c73f9b384b54f081ac1efd7a2fc896094351434a395b6b7cc11c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:35.936974Z","signature_b64":"0R4OXQuJA8TY6hi6hmWyx9RqwPCKJXzPKUmcR1BCgfNHLsz84EMskleuizVD/eCoCm5yFSkGHXPJ5PPVezDnCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9a026c46da7733dff7585380673124e393edd6bbba36e06ade8bb87014fef2ba","last_reissued_at":"2026-07-05T11:18:35.936447Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:35.936447Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cost-Optimal Active AI Model Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adam Fisch, Alekh Agarwal, Anastasios N. Angelopoulos, Jacob Eisenstein, Jonathan Berant","submitted_at":"2025-06-09T17:14:41Z","abstract_excerpt":"The development lifecycle of generative AI systems requires continual evaluation, data acquisition, and annotation, which is costly in both resources and time. In practice, rapid iteration often makes it necessary to rely on synthetic annotation data because of the low cost, despite the potential for substantial bias. In this paper, we develop novel, cost-aware methods for actively balancing the use of a cheap, but often inaccurate, weak rater -- such as a model-based autorater that is designed to automatically assess the quality of generated content -- with a more expensive, but also more acc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07949","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07949/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07949","created_at":"2026-07-05T11:18:35.936514+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07949v1","created_at":"2026-07-05T11:18:35.936514+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07949","created_at":"2026-07-05T11:18:35.936514+00:00"},{"alias_kind":"pith_short_12","alias_value":"TIBGYRW2O4Z5","created_at":"2026-07-05T11:18:35.936514+00:00"},{"alias_kind":"pith_short_16","alias_value":"TIBGYRW2O4Z5752Y","created_at":"2026-07-05T11:18:35.936514+00:00"},{"alias_kind":"pith_short_8","alias_value":"TIBGYRW2","created_at":"2026-07-05T11:18:35.936514+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08347","citing_title":"Prediction-Powered Active Testing","ref_index":70,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03211","citing_title":"Optimized Labeling Resource Allocation for Prediction-Assisted Inference via OPAL","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25998","citing_title":"Causal methods for LLM development and evaluation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2601.20251","citing_title":"Efficient Evaluation of LLM Performance with Statistical Guarantees","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11638","citing_title":"Learning U-Statistics with Active Inference","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05973","citing_title":"Towards Reliable LLM Evaluation: Correcting the Winner's Curse in Adaptive Benchmarking","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O","json":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O.json","graph_json":"https://pith.science/api/pith-number/TIBGYRW2O4Z5752YKOAGOMJE4O/graph.json","events_json":"https://pith.science/api/pith-number/TIBGYRW2O4Z5752YKOAGOMJE4O/events.json","paper":"https://pith.science/paper/TIBGYRW2"},"agent_actions":{"view_html":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O","download_json":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O.json","view_paper":"https://pith.science/paper/TIBGYRW2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07949&json=true","fetch_graph":"https://pith.science/api/pith-number/TIBGYRW2O4Z5752YKOAGOMJE4O/graph.json","fetch_events":"https://pith.science/api/pith-number/TIBGYRW2O4Z5752YKOAGOMJE4O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O/action/storage_attestation","attest_author":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O/action/author_attestation","sign_citation":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O/action/citation_signature","submit_replication":"https://pith.science/pith/TIBGYRW2O4Z5752YKOAGOMJE4O/action/replication_record"}},"created_at":"2026-07-05T11:18:35.936514+00:00","updated_at":"2026-07-05T11:18:35.936514+00:00"}