{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4KJ7UYLWYJFX73QRJ3U324P3PR","short_pith_number":"pith:4KJ7UYLW","schema_version":"1.0","canonical_sha256":"e293fa6176c24b7fee114ee9bd71fb7c4ef649cfed0954494648d73ab706b08d","source":{"kind":"arxiv","id":"2312.02368","version":1},"attestation_state":"computed","paper":{"title":"RINAS: Training with Dataset Shuffling Can Be General and Fast","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC","cs.LG","cs.PF"],"primary_cat":"cs.DB","authors_text":"Geoffrey Fox, Jiechen Zhao, Qiang Su, Tianle Zhong, Xindi Guo","submitted_at":"2023-12-04T21:50:08Z","abstract_excerpt":"Deep learning datasets are expanding at an unprecedented pace, creating new challenges for data processing in model training pipelines. A crucial aspect of these pipelines is dataset shuffling, which significantly improves unbiased learning and convergence accuracy by adhering to the principles of random sampling. However, loading shuffled data for large datasets incurs significant overhead in the deep learning pipeline and severely impacts the end-to-end training throughput. To mitigate this, current deep learning systems often resort to partial dataset shuffling, sacrificing global randomnes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02368","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DB","submitted_at":"2023-12-04T21:50:08Z","cross_cats_sorted":["cs.DC","cs.LG","cs.PF"],"title_canon_sha256":"8408c3b7bb25efd5188a60b979fbff9366ae6ceb7f480889c7883d9616db356f","abstract_canon_sha256":"6e1779b24f5c6b4208dc1904f4235383dbee6debc868bd2725f5a29fe57e3783"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:20:17.159851Z","signature_b64":"oEKuXyrZc7BEwABptdfOSnbvAiNF9e/w1TAUU9V7zOXwbqSLEzEO/GI9RJkyYoyBkEnX0onBJviWFz2CritVBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e293fa6176c24b7fee114ee9bd71fb7c4ef649cfed0954494648d73ab706b08d","last_reissued_at":"2026-07-05T07:20:17.159414Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:20:17.159414Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RINAS: Training with Dataset Shuffling Can Be General and Fast","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC","cs.LG","cs.PF"],"primary_cat":"cs.DB","authors_text":"Geoffrey Fox, Jiechen Zhao, Qiang Su, Tianle Zhong, Xindi Guo","submitted_at":"2023-12-04T21:50:08Z","abstract_excerpt":"Deep learning datasets are expanding at an unprecedented pace, creating new challenges for data processing in model training pipelines. A crucial aspect of these pipelines is dataset shuffling, which significantly improves unbiased learning and convergence accuracy by adhering to the principles of random sampling. However, loading shuffled data for large datasets incurs significant overhead in the deep learning pipeline and severely impacts the end-to-end training throughput. To mitigate this, current deep learning systems often resort to partial dataset shuffling, sacrificing global randomnes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02368","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02368/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02368","created_at":"2026-07-05T07:20:17.159486+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02368v1","created_at":"2026-07-05T07:20:17.159486+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02368","created_at":"2026-07-05T07:20:17.159486+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KJ7UYLWYJFX","created_at":"2026-07-05T07:20:17.159486+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KJ7UYLWYJFX73QR","created_at":"2026-07-05T07:20:17.159486+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KJ7UYLW","created_at":"2026-07-05T07:20:17.159486+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.10153","citing_title":"EVOS: Efficient Implicit Neural Training via EVOlutionary Selector","ref_index":66,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR","json":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR.json","graph_json":"https://pith.science/api/pith-number/4KJ7UYLWYJFX73QRJ3U324P3PR/graph.json","events_json":"https://pith.science/api/pith-number/4KJ7UYLWYJFX73QRJ3U324P3PR/events.json","paper":"https://pith.science/paper/4KJ7UYLW"},"agent_actions":{"view_html":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR","download_json":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR.json","view_paper":"https://pith.science/paper/4KJ7UYLW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02368&json=true","fetch_graph":"https://pith.science/api/pith-number/4KJ7UYLWYJFX73QRJ3U324P3PR/graph.json","fetch_events":"https://pith.science/api/pith-number/4KJ7UYLWYJFX73QRJ3U324P3PR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR/action/storage_attestation","attest_author":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR/action/author_attestation","sign_citation":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR/action/citation_signature","submit_replication":"https://pith.science/pith/4KJ7UYLWYJFX73QRJ3U324P3PR/action/replication_record"}},"created_at":"2026-07-05T07:20:17.159486+00:00","updated_at":"2026-07-05T07:20:17.159486+00:00"}