{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VHWVEF5TQ52OVT3N3DJC6VC6D5","short_pith_number":"pith:VHWVEF5T","schema_version":"1.0","canonical_sha256":"a9ed5217b38774eacf6dd8d22f545e1f50b2dd29e611aeffbc497c66dfccc5ed","source":{"kind":"arxiv","id":"2504.16379","version":1},"attestation_state":"computed","paper":{"title":"SplitReason: Learning To Offload Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ahmed F. AbouElhamayed, Anthony Fei, Chi-Chih Chang, Mohamed S. Abdelfattah, Yash Akhauri, Yueying Li","submitted_at":"2025-04-23T03:00:02Z","abstract_excerpt":"Reasoning in large language models (LLMs) tends to produce substantially longer token generation sequences than simpler language modeling tasks. This extended generation length reflects the multi-step, compositional nature of reasoning and is often correlated with higher solution accuracy. From an efficiency perspective, longer token generation exacerbates the inherently sequential and memory-bound decoding phase of LLMs. However, not all parts of this expensive reasoning process are equally difficult to generate. We leverage this observation by offloading only the most challenging parts of th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.16379","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-23T03:00:02Z","cross_cats_sorted":[],"title_canon_sha256":"8270d19d6b50c9b72acb404bd5fec03db0ebc31bff7fd45f49a213cb01e8804a","abstract_canon_sha256":"f1ea988f82200bd00244f51774b4e7667f4986872bf5dd969ce94bd4c2d4d9f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:45.911239Z","signature_b64":"6fpV3zEZEvUKP2h40bpzwEYJ6Si5ZntT9PUbFsZXCvlW2rVx42VvgSJutx0akvmBDyteBVcqaUY/l795NY/fAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a9ed5217b38774eacf6dd8d22f545e1f50b2dd29e611aeffbc497c66dfccc5ed","last_reissued_at":"2026-07-05T10:52:45.910772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:45.910772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SplitReason: Learning To Offload Reasoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ahmed F. AbouElhamayed, Anthony Fei, Chi-Chih Chang, Mohamed S. Abdelfattah, Yash Akhauri, Yueying Li","submitted_at":"2025-04-23T03:00:02Z","abstract_excerpt":"Reasoning in large language models (LLMs) tends to produce substantially longer token generation sequences than simpler language modeling tasks. This extended generation length reflects the multi-step, compositional nature of reasoning and is often correlated with higher solution accuracy. From an efficiency perspective, longer token generation exacerbates the inherently sequential and memory-bound decoding phase of LLMs. However, not all parts of this expensive reasoning process are equally difficult to generate. We leverage this observation by offloading only the most challenging parts of th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.16379","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.16379/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.16379","created_at":"2026-07-05T10:52:45.910830+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.16379v1","created_at":"2026-07-05T10:52:45.910830+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.16379","created_at":"2026-07-05T10:52:45.910830+00:00"},{"alias_kind":"pith_short_12","alias_value":"VHWVEF5TQ52O","created_at":"2026-07-05T10:52:45.910830+00:00"},{"alias_kind":"pith_short_16","alias_value":"VHWVEF5TQ52OVT3N","created_at":"2026-07-05T10:52:45.910830+00:00"},{"alias_kind":"pith_short_8","alias_value":"VHWVEF5T","created_at":"2026-07-05T10:52:45.910830+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02011","citing_title":"Extreme Low-Bit Inference in Reasoning Models: Failure Modes and Targeted Recovery","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16630","citing_title":"PrivScope: Task-scoped Disclosure Control for Hybrid Agentic Systems","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18839","citing_title":"One Step Forward and K Steps Back: Better Reasoning with Denoising Recursion Models","ref_index":141,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5","json":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5.json","graph_json":"https://pith.science/api/pith-number/VHWVEF5TQ52OVT3N3DJC6VC6D5/graph.json","events_json":"https://pith.science/api/pith-number/VHWVEF5TQ52OVT3N3DJC6VC6D5/events.json","paper":"https://pith.science/paper/VHWVEF5T"},"agent_actions":{"view_html":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5","download_json":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5.json","view_paper":"https://pith.science/paper/VHWVEF5T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.16379&json=true","fetch_graph":"https://pith.science/api/pith-number/VHWVEF5TQ52OVT3N3DJC6VC6D5/graph.json","fetch_events":"https://pith.science/api/pith-number/VHWVEF5TQ52OVT3N3DJC6VC6D5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5/action/storage_attestation","attest_author":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5/action/author_attestation","sign_citation":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5/action/citation_signature","submit_replication":"https://pith.science/pith/VHWVEF5TQ52OVT3N3DJC6VC6D5/action/replication_record"}},"created_at":"2026-07-05T10:52:45.910830+00:00","updated_at":"2026-07-05T10:52:45.910830+00:00"}