{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:UHKNGPOK2EQMH2IXTLDMJ67AOK","short_pith_number":"pith:UHKNGPOK","schema_version":"1.0","canonical_sha256":"a1d4d33dcad120c3e9179ac6c4fbe0728ae3cb88e0ad59ec9afdfc0698802e26","source":{"kind":"arxiv","id":"2601.16956","version":1},"attestation_state":"computed","paper":{"title":"DataStates-LLM: Scalable Checkpointing for Transformer Models Using Composable State Providers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.PF"],"primary_cat":"cs.DC","authors_text":"Avinash Maurya, Bogdan Nicolae, Franck Cappello, M. Mustafa Rafique","submitted_at":"2026-01-23T18:26:14Z","abstract_excerpt":"The rapid growth of Large Transformer-based models, specifically Large Language Models (LLMs), now scaling to trillions of parameters, has necessitated training across thousands of GPUs using complex hybrid parallelism strategies (e.g., data, tensor, and pipeline parallelism). Checkpointing this massive, distributed state is critical for a wide range of use cases, such as resilience, suspend-resume, investigating undesirable training trajectories, and explaining model evolution. However, existing checkpointing solutions typically treat model state as opaque binary blobs, ignoring the ``3D hete"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.16956","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2026-01-23T18:26:14Z","cross_cats_sorted":["cs.AI","cs.PF"],"title_canon_sha256":"78e2d216bdb60fc96de45c002375ec7821a3f98b372b90d6e6a444c48d4a6ea9","abstract_canon_sha256":"6646017373afb669f8ea340bfd075ef37e20a87c5bba671222e19e79ecaed7c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-29T00:14:01.899358Z","signature_b64":"sxlq40tk682ZXvknhlwZBGhsiOjNvOpgf0dzXD9Ls06DwSyn6kH+xhwnf2V6h6EhUe1MoZTLhX4Wdnm0C8bXBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a1d4d33dcad120c3e9179ac6c4fbe0728ae3cb88e0ad59ec9afdfc0698802e26","last_reissued_at":"2026-06-29T00:14:01.898779Z","signature_status":"signed_v1","first_computed_at":"2026-06-29T00:14:01.898779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DataStates-LLM: Scalable Checkpointing for Transformer Models Using Composable State Providers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.PF"],"primary_cat":"cs.DC","authors_text":"Avinash Maurya, Bogdan Nicolae, Franck Cappello, M. Mustafa Rafique","submitted_at":"2026-01-23T18:26:14Z","abstract_excerpt":"The rapid growth of Large Transformer-based models, specifically Large Language Models (LLMs), now scaling to trillions of parameters, has necessitated training across thousands of GPUs using complex hybrid parallelism strategies (e.g., data, tensor, and pipeline parallelism). Checkpointing this massive, distributed state is critical for a wide range of use cases, such as resilience, suspend-resume, investigating undesirable training trajectories, and explaining model evolution. However, existing checkpointing solutions typically treat model state as opaque binary blobs, ignoring the ``3D hete"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.16956","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.16956/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.16956","created_at":"2026-06-29T00:14:01.898845+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.16956v1","created_at":"2026-06-29T00:14:01.898845+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.16956","created_at":"2026-06-29T00:14:01.898845+00:00"},{"alias_kind":"pith_short_12","alias_value":"UHKNGPOK2EQM","created_at":"2026-06-29T00:14:01.898845+00:00"},{"alias_kind":"pith_short_16","alias_value":"UHKNGPOK2EQMH2IX","created_at":"2026-06-29T00:14:01.898845+00:00"},{"alias_kind":"pith_short_8","alias_value":"UHKNGPOK","created_at":"2026-06-29T00:14:01.898845+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2605.11215","citing_title":"ReCoVer: Resilient LLM Pre-Training System via Fault-Tolerant Collective and Versatile Workload","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11215","citing_title":"ReCoVer: Resilient LLM Pre-Training System via Fault-Tolerant Collective and Versatile Workload","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK","json":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK.json","graph_json":"https://pith.science/api/pith-number/UHKNGPOK2EQMH2IXTLDMJ67AOK/graph.json","events_json":"https://pith.science/api/pith-number/UHKNGPOK2EQMH2IXTLDMJ67AOK/events.json","paper":"https://pith.science/paper/UHKNGPOK"},"agent_actions":{"view_html":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK","download_json":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK.json","view_paper":"https://pith.science/paper/UHKNGPOK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.16956&json=true","fetch_graph":"https://pith.science/api/pith-number/UHKNGPOK2EQMH2IXTLDMJ67AOK/graph.json","fetch_events":"https://pith.science/api/pith-number/UHKNGPOK2EQMH2IXTLDMJ67AOK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK/action/storage_attestation","attest_author":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK/action/author_attestation","sign_citation":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK/action/citation_signature","submit_replication":"https://pith.science/pith/UHKNGPOK2EQMH2IXTLDMJ67AOK/action/replication_record"}},"created_at":"2026-06-29T00:14:01.898845+00:00","updated_at":"2026-06-29T00:14:01.898845+00:00"}