{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:V6EP7UZZC4FU6T3LR57IU5TZ7B","short_pith_number":"pith:V6EP7UZZ","schema_version":"1.0","canonical_sha256":"af88ffd339170b4f4f6b8f7e8a7679f84f25ba16f0d746fc6cab67c5d9d87515","source":{"kind":"arxiv","id":"2506.00482","version":1},"attestation_state":"computed","paper":{"title":"BenchHub: A Unified Benchmark Suite for Holistic and Customizable LLM Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alice Oh, Amit Agarwal, Eunsu Kim, Guijin Son, Haneul Yoo, Hitesh Patel","submitted_at":"2025-05-31T09:24:32Z","abstract_excerpt":"As large language models (LLMs) continue to advance, the need for up-to-date and well-organized benchmarks becomes increasingly critical. However, many existing datasets are scattered, difficult to manage, and make it challenging to perform evaluations tailored to specific needs or domains, despite the growing importance of domain-specific models in areas such as math or code. In this paper, we introduce BenchHub, a dynamic benchmark repository that empowers researchers and developers to evaluate LLMs more effectively. BenchHub aggregates and automatically classifies benchmark datasets from di"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00482","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-31T09:24:32Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d701918ed4b18d39efc7d55bb71dd091566629322749c2fd66436d92f65d126d","abstract_canon_sha256":"a8190040029976d7c4949fd225347ab936eca5e3cfef44bd27ed9ec22eef5aa8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:35.701956Z","signature_b64":"kewVFpbTL6TlkR6bHYCq+bQsf+aTIQstxLMdu0OiGr3hpFu/VXrQY7N3IH6yjO+1YjZDQOSsc2XdW2KRi9bZBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"af88ffd339170b4f4f6b8f7e8a7679f84f25ba16f0d746fc6cab67c5d9d87515","last_reissued_at":"2026-07-05T11:13:35.701596Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:35.701596Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BenchHub: A Unified Benchmark Suite for Holistic and Customizable LLM Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alice Oh, Amit Agarwal, Eunsu Kim, Guijin Son, Haneul Yoo, Hitesh Patel","submitted_at":"2025-05-31T09:24:32Z","abstract_excerpt":"As large language models (LLMs) continue to advance, the need for up-to-date and well-organized benchmarks becomes increasingly critical. However, many existing datasets are scattered, difficult to manage, and make it challenging to perform evaluations tailored to specific needs or domains, despite the growing importance of domain-specific models in areas such as math or code. In this paper, we introduce BenchHub, a dynamic benchmark repository that empowers researchers and developers to evaluate LLMs more effectively. BenchHub aggregates and automatically classifies benchmark datasets from di"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00482","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00482/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00482","created_at":"2026-07-05T11:13:35.701651+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00482v1","created_at":"2026-07-05T11:13:35.701651+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00482","created_at":"2026-07-05T11:13:35.701651+00:00"},{"alias_kind":"pith_short_12","alias_value":"V6EP7UZZC4FU","created_at":"2026-07-05T11:13:35.701651+00:00"},{"alias_kind":"pith_short_16","alias_value":"V6EP7UZZC4FU6T3L","created_at":"2026-07-05T11:13:35.701651+00:00"},{"alias_kind":"pith_short_8","alias_value":"V6EP7UZZ","created_at":"2026-07-05T11:13:35.701651+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B","json":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B.json","graph_json":"https://pith.science/api/pith-number/V6EP7UZZC4FU6T3LR57IU5TZ7B/graph.json","events_json":"https://pith.science/api/pith-number/V6EP7UZZC4FU6T3LR57IU5TZ7B/events.json","paper":"https://pith.science/paper/V6EP7UZZ"},"agent_actions":{"view_html":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B","download_json":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B.json","view_paper":"https://pith.science/paper/V6EP7UZZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00482&json=true","fetch_graph":"https://pith.science/api/pith-number/V6EP7UZZC4FU6T3LR57IU5TZ7B/graph.json","fetch_events":"https://pith.science/api/pith-number/V6EP7UZZC4FU6T3LR57IU5TZ7B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B/action/storage_attestation","attest_author":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B/action/author_attestation","sign_citation":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B/action/citation_signature","submit_replication":"https://pith.science/pith/V6EP7UZZC4FU6T3LR57IU5TZ7B/action/replication_record"}},"created_at":"2026-07-05T11:13:35.701651+00:00","updated_at":"2026-07-05T11:13:35.701651+00:00"}