{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6GKIUJ46J5HIMTCRWQYOEH4PBG","short_pith_number":"pith:6GKIUJ46","schema_version":"1.0","canonical_sha256":"f1948a279e4f4e864c51b430e21f8f09a8661c1f623ab15fe0887b348d6d98e2","source":{"kind":"arxiv","id":"2506.15961","version":2},"attestation_state":"computed","paper":{"title":"TrainVerify: Equivalence-Based Verification for Distributed LLM Training","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Cheng Tan, Fan Yang, Peng Huang, Xian Zhang, Yi Zhu, Youshan Miao, Yunchi Lu","submitted_at":"2025-06-19T02:10:06Z","abstract_excerpt":"Training large language models (LLMs) at scale requires parallel execution across thousands of devices, incurring enormous computational costs. Yet, these costly distributed trainings are rarely verified, leaving them prone to silent errors and potentially wasting millions of GPU hours. We introduce TrainVerify, a system for verifiable distributed training of LLMs. Given a deep learning model's logical specification as the ground truth, TrainVerify formally verifies that a distributed parallel execution plan is mathematically equivalent to it. Direct verification is notoriously difficult due t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.15961","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.DC","submitted_at":"2025-06-19T02:10:06Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"00bf2377af6fd17c8dba0db09b472cfeacbfbcb8587b304ce55f599ce775d247","abstract_canon_sha256":"e79a1304102d39906052a1611ad9047bfeaa4aa65721a06a6c1e5497af0db639"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:05.723715Z","signature_b64":"307N6FCXFIiTJ6liDhnH2ezo6Oa3kJVN/LbiA5IE6Cu3MQHVgBlzYOcWikepi4O6pHOXyevjj6xDpefn+0JABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f1948a279e4f4e864c51b430e21f8f09a8661c1f623ab15fe0887b348d6d98e2","last_reissued_at":"2026-07-05T11:26:05.723103Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:05.723103Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TrainVerify: Equivalence-Based Verification for Distributed LLM Training","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.DC","authors_text":"Cheng Tan, Fan Yang, Peng Huang, Xian Zhang, Yi Zhu, Youshan Miao, Yunchi Lu","submitted_at":"2025-06-19T02:10:06Z","abstract_excerpt":"Training large language models (LLMs) at scale requires parallel execution across thousands of devices, incurring enormous computational costs. Yet, these costly distributed trainings are rarely verified, leaving them prone to silent errors and potentially wasting millions of GPU hours. We introduce TrainVerify, a system for verifiable distributed training of LLMs. Given a deep learning model's logical specification as the ground truth, TrainVerify formally verifies that a distributed parallel execution plan is mathematically equivalent to it. Direct verification is notoriously difficult due t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.15961","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.15961/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.15961","created_at":"2026-07-05T11:26:05.723188+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.15961v2","created_at":"2026-07-05T11:26:05.723188+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.15961","created_at":"2026-07-05T11:26:05.723188+00:00"},{"alias_kind":"pith_short_12","alias_value":"6GKIUJ46J5HI","created_at":"2026-07-05T11:26:05.723188+00:00"},{"alias_kind":"pith_short_16","alias_value":"6GKIUJ46J5HIMTCR","created_at":"2026-07-05T11:26:05.723188+00:00"},{"alias_kind":"pith_short_8","alias_value":"6GKIUJ46","created_at":"2026-07-05T11:26:05.723188+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2509.10694","citing_title":"Verifying Computational Graphs in Production-Grade Distributed Machine Learning Frameworks","ref_index":58,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG","json":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG.json","graph_json":"https://pith.science/api/pith-number/6GKIUJ46J5HIMTCRWQYOEH4PBG/graph.json","events_json":"https://pith.science/api/pith-number/6GKIUJ46J5HIMTCRWQYOEH4PBG/events.json","paper":"https://pith.science/paper/6GKIUJ46"},"agent_actions":{"view_html":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG","download_json":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG.json","view_paper":"https://pith.science/paper/6GKIUJ46","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.15961&json=true","fetch_graph":"https://pith.science/api/pith-number/6GKIUJ46J5HIMTCRWQYOEH4PBG/graph.json","fetch_events":"https://pith.science/api/pith-number/6GKIUJ46J5HIMTCRWQYOEH4PBG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG/action/storage_attestation","attest_author":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG/action/author_attestation","sign_citation":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG/action/citation_signature","submit_replication":"https://pith.science/pith/6GKIUJ46J5HIMTCRWQYOEH4PBG/action/replication_record"}},"created_at":"2026-07-05T11:26:05.723188+00:00","updated_at":"2026-07-05T11:26:05.723188+00:00"}