{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5LTK4BL3XNWEMHHYS4LRD2PXSR","short_pith_number":"pith:5LTK4BL3","schema_version":"1.0","canonical_sha256":"eae6ae057bbb6c461cf8971711e9f79464165acfd2a0dd3b4ff28e89dc87ae04","source":{"kind":"arxiv","id":"2402.00861","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Large Language Models for Generalization and Robustness via Data Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenghua Lin, Frank Guerin, Yucheng Li, Yunhao Guo","submitted_at":"2024-02-01T18:56:18Z","abstract_excerpt":"Existing methods for evaluating large language models face challenges such as data contamination, sensitivity to prompts, and the high cost of benchmark creation. To address this, we propose a lossless data compression based evaluation approach that tests how models' predictive abilities generalize after their training cutoff. Specifically, we collect comprehensive test data spanning 83 months from 2017 to 2023 and split the data into training and testing periods according to models' training data cutoff. We measure: 1) the compression performance on the testing period as a measure of generali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00861","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-01T18:56:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fcbbec2ebc7f937dfce6cd28d4f048df6195f5a5b551c0c6cbca160855bf8913","abstract_canon_sha256":"b9bcbcdae08c3c61bbd2df5de11715af35c8bcf35ffea0d323d23c4c5b6a0edf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:03.977395Z","signature_b64":"zr7nOnXiIGp8ReVEn0K6n74v9OshBKrt5YkHIYIbu44FHZsqzWJ3MFsiVkrFGXLBNX9qspuCPrSdwD7HTRG5CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eae6ae057bbb6c461cf8971711e9f79464165acfd2a0dd3b4ff28e89dc87ae04","last_reissued_at":"2026-07-05T07:41:03.976964Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:03.976964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Large Language Models for Generalization and Robustness via Data Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Chenghua Lin, Frank Guerin, Yucheng Li, Yunhao Guo","submitted_at":"2024-02-01T18:56:18Z","abstract_excerpt":"Existing methods for evaluating large language models face challenges such as data contamination, sensitivity to prompts, and the high cost of benchmark creation. To address this, we propose a lossless data compression based evaluation approach that tests how models' predictive abilities generalize after their training cutoff. Specifically, we collect comprehensive test data spanning 83 months from 2017 to 2023 and split the data into training and testing periods according to models' training data cutoff. We measure: 1) the compression performance on the testing period as a measure of generali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00861","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00861/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00861","created_at":"2026-07-05T07:41:03.977021+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00861v2","created_at":"2026-07-05T07:41:03.977021+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00861","created_at":"2026-07-05T07:41:03.977021+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LTK4BL3XNWE","created_at":"2026-07-05T07:41:03.977021+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LTK4BL3XNWEMHHY","created_at":"2026-07-05T07:41:03.977021+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LTK4BL3","created_at":"2026-07-05T07:41:03.977021+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR","json":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR.json","graph_json":"https://pith.science/api/pith-number/5LTK4BL3XNWEMHHYS4LRD2PXSR/graph.json","events_json":"https://pith.science/api/pith-number/5LTK4BL3XNWEMHHYS4LRD2PXSR/events.json","paper":"https://pith.science/paper/5LTK4BL3"},"agent_actions":{"view_html":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR","download_json":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR.json","view_paper":"https://pith.science/paper/5LTK4BL3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00861&json=true","fetch_graph":"https://pith.science/api/pith-number/5LTK4BL3XNWEMHHYS4LRD2PXSR/graph.json","fetch_events":"https://pith.science/api/pith-number/5LTK4BL3XNWEMHHYS4LRD2PXSR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR/action/storage_attestation","attest_author":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR/action/author_attestation","sign_citation":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR/action/citation_signature","submit_replication":"https://pith.science/pith/5LTK4BL3XNWEMHHYS4LRD2PXSR/action/replication_record"}},"created_at":"2026-07-05T07:41:03.977021+00:00","updated_at":"2026-07-05T07:41:03.977021+00:00"}