{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S22CP3GECNKYKLBRRRWEVRFH5Y","short_pith_number":"pith:S22CP3GE","schema_version":"1.0","canonical_sha256":"96b427ecc41355852c318c6c4ac4a7ee2ad811a82fe5f1c7bb374a3880646a67","source":{"kind":"arxiv","id":"2406.17681","version":2},"attestation_state":"computed","paper":{"title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claudia Tang, Kun Qian, Maximillian Chen, Shunji Wan, Xuanming Zhang, Youzhi Wang, Zhou Yu","submitted_at":"2024-06-25T16:13:53Z","abstract_excerpt":"As large language models achieve impressive scores on traditional benchmarks, an increasing number of researchers are becoming concerned about benchmark data leakage during pre-training, commonly known as the data contamination problem. To ensure fair evaluation, recent benchmarks release only the training and validation sets, keeping the test set labels closed-source. They require anyone wishing to evaluate his language model to submit the model's predictions for centralized processing and then publish the model's result on their leaderboard. However, this submission process is inefficient an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.17681","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-25T16:13:53Z","cross_cats_sorted":[],"title_canon_sha256":"435ff44ee88e89f6b6a29c16d9cbe883f985762c9878acac60d4f8c73a6fbd7f","abstract_canon_sha256":"879535f07ff1e7c9997607bbbbf9339dcfaff121b6e0348d7a304343edb236bc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:07.534091Z","signature_b64":"lKrpclq0N7DxKO053E87G78iTXllWGBnMIDHJZAx0qcND4msi4phyk1ygAC4X+O2TmMaWJSX4FoE2Ob4lHx9AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96b427ecc41355852c318c6c4ac4a7ee2ad811a82fe5f1c7bb374a3880646a67","last_reissued_at":"2026-07-05T08:37:07.533625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:07.533625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Claudia Tang, Kun Qian, Maximillian Chen, Shunji Wan, Xuanming Zhang, Youzhi Wang, Zhou Yu","submitted_at":"2024-06-25T16:13:53Z","abstract_excerpt":"As large language models achieve impressive scores on traditional benchmarks, an increasing number of researchers are becoming concerned about benchmark data leakage during pre-training, commonly known as the data contamination problem. To ensure fair evaluation, recent benchmarks release only the training and validation sets, keeping the test set labels closed-source. They require anyone wishing to evaluate his language model to submit the model's predictions for centralized processing and then publish the model's result on their leaderboard. However, this submission process is inefficient an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.17681","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.17681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.17681","created_at":"2026-07-05T08:37:07.533683+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.17681v2","created_at":"2026-07-05T08:37:07.533683+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.17681","created_at":"2026-07-05T08:37:07.533683+00:00"},{"alias_kind":"pith_short_12","alias_value":"S22CP3GECNKY","created_at":"2026-07-05T08:37:07.533683+00:00"},{"alias_kind":"pith_short_16","alias_value":"S22CP3GECNKYKLBR","created_at":"2026-07-05T08:37:07.533683+00:00"},{"alias_kind":"pith_short_8","alias_value":"S22CP3GE","created_at":"2026-07-05T08:37:07.533683+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26133","citing_title":"Pretraining Data Exposure in Large Language Models: A Survey of Membership Inference, Data Contamination, and Security Implications","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17829","citing_title":"Interactive Evaluation Requires a Design Science","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19035","citing_title":"Investigating Advanced Reasoning of Large Language Models via Black-Box Environment Interaction","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y","json":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y.json","graph_json":"https://pith.science/api/pith-number/S22CP3GECNKYKLBRRRWEVRFH5Y/graph.json","events_json":"https://pith.science/api/pith-number/S22CP3GECNKYKLBRRRWEVRFH5Y/events.json","paper":"https://pith.science/paper/S22CP3GE"},"agent_actions":{"view_html":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y","download_json":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y.json","view_paper":"https://pith.science/paper/S22CP3GE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.17681&json=true","fetch_graph":"https://pith.science/api/pith-number/S22CP3GECNKYKLBRRRWEVRFH5Y/graph.json","fetch_events":"https://pith.science/api/pith-number/S22CP3GECNKYKLBRRRWEVRFH5Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y/action/storage_attestation","attest_author":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y/action/author_attestation","sign_citation":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y/action/citation_signature","submit_replication":"https://pith.science/pith/S22CP3GECNKYKLBRRRWEVRFH5Y/action/replication_record"}},"created_at":"2026-07-05T08:37:07.533683+00:00","updated_at":"2026-07-05T08:37:07.533683+00:00"}