{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:EOR3QYZX3JPA3JWMOR6INNJ2GN","short_pith_number":"pith:EOR3QYZX","schema_version":"1.0","canonical_sha256":"23a3b86337da5e0da6cc747c86b53a33622600298c7f6869bc89ba6a3b66ef00","source":{"kind":"arxiv","id":"2506.03077","version":1},"attestation_state":"computed","paper":{"title":"StreamBP: Memory-Efficient Exact Backpropagation for Long Sequence Training of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Lei Zhao, Mengqi Li, Qijun Luo, Xiao Li","submitted_at":"2025-06-03T16:54:15Z","abstract_excerpt":"Training language models on long sequence data is a demanding requirement for enhancing the model's capability on complex tasks, e.g., long-chain reasoning. However, as the sequence length scales up, the memory cost for storing activation values becomes huge during the Backpropagation (BP) process, even with the application of gradient checkpointing technique. To tackle this challenge, we propose a memory-efficient and exact BP method called StreamBP, which performs a linear decomposition of the chain rule along the sequence dimension in a layer-wise manner, significantly reducing the memory c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03077","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-03T16:54:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f7c434e5c4ccd0a3271918b5d5352a41000294747368316c4c6fad68bab76f40","abstract_canon_sha256":"65857c4246725fd00fda5bae5ff58eab5081097a95d048b3b8629c27d2f2bbe7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:14.482238Z","signature_b64":"DTAs2Ut/eL2HPSTlqSXIQDCs+TVfE/qugZk+3hXv53StKz/51ExOO5owP9FHHiJpaF1BFTBT2NRF8VhSqFSrDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23a3b86337da5e0da6cc747c86b53a33622600298c7f6869bc89ba6a3b66ef00","last_reissued_at":"2026-07-05T11:15:14.481734Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:14.481734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"StreamBP: Memory-Efficient Exact Backpropagation for Long Sequence Training of LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Lei Zhao, Mengqi Li, Qijun Luo, Xiao Li","submitted_at":"2025-06-03T16:54:15Z","abstract_excerpt":"Training language models on long sequence data is a demanding requirement for enhancing the model's capability on complex tasks, e.g., long-chain reasoning. However, as the sequence length scales up, the memory cost for storing activation values becomes huge during the Backpropagation (BP) process, even with the application of gradient checkpointing technique. To tackle this challenge, we propose a memory-efficient and exact BP method called StreamBP, which performs a linear decomposition of the chain rule along the sequence dimension in a layer-wise manner, significantly reducing the memory c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03077","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03077/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03077","created_at":"2026-07-05T11:15:14.481806+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03077v1","created_at":"2026-07-05T11:15:14.481806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03077","created_at":"2026-07-05T11:15:14.481806+00:00"},{"alias_kind":"pith_short_12","alias_value":"EOR3QYZX3JPA","created_at":"2026-07-05T11:15:14.481806+00:00"},{"alias_kind":"pith_short_16","alias_value":"EOR3QYZX3JPA3JWM","created_at":"2026-07-05T11:15:14.481806+00:00"},{"alias_kind":"pith_short_8","alias_value":"EOR3QYZX","created_at":"2026-07-05T11:15:14.481806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16154","citing_title":"Learn Where Outcomes Diverge: Efficient VLA RL via Probabilistic Chunk Masking","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN","json":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN.json","graph_json":"https://pith.science/api/pith-number/EOR3QYZX3JPA3JWMOR6INNJ2GN/graph.json","events_json":"https://pith.science/api/pith-number/EOR3QYZX3JPA3JWMOR6INNJ2GN/events.json","paper":"https://pith.science/paper/EOR3QYZX"},"agent_actions":{"view_html":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN","download_json":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN.json","view_paper":"https://pith.science/paper/EOR3QYZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03077&json=true","fetch_graph":"https://pith.science/api/pith-number/EOR3QYZX3JPA3JWMOR6INNJ2GN/graph.json","fetch_events":"https://pith.science/api/pith-number/EOR3QYZX3JPA3JWMOR6INNJ2GN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN/action/storage_attestation","attest_author":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN/action/author_attestation","sign_citation":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN/action/citation_signature","submit_replication":"https://pith.science/pith/EOR3QYZX3JPA3JWMOR6INNJ2GN/action/replication_record"}},"created_at":"2026-07-05T11:15:14.481806+00:00","updated_at":"2026-07-05T11:15:14.481806+00:00"}