{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MSB6AUTIQ5ROXJ2CHFYLKGPWU3","short_pith_number":"pith:MSB6AUTI","schema_version":"1.0","canonical_sha256":"6483e052688762eba7423970b519f6a6c1c357b95b85dae440e0f5a30bbc794a","source":{"kind":"arxiv","id":"2508.10026","version":1},"attestation_state":"computed","paper":{"title":"SABER: Switchable and Balanced Training for Efficient LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiaming Song, Kai Zhao, Lusheng Zhang, Qiang Zhang, Shien He, Tianjiao Li, Yanjun Zhao","submitted_at":"2025-08-08T11:27:48Z","abstract_excerpt":"Large language models (LLMs) empowered by chain-of-thought reasoning have achieved impressive accuracy on complex tasks but suffer from excessive inference costs and latency when applied uniformly to all problems. We propose SABER (Switchable and Balanced Training for Efficient LLM Reasoning), a reinforcement learning framework that endows LLMs with user-controllable, token-budgeted reasoning. SABER first profiles each training example's base-model thinking token usage and assigns it to one of the predefined budget tiers. During fine-tuning, the model is guided by system prompts and length-awa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.10026","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-08-08T11:27:48Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b233078497eeb01bc3bc013c7aa4288752416e2d02e5706d087cab3e77e1bdd1","abstract_canon_sha256":"5455237eed5a56c110605b346118348a4db420fd087d728d518154caf1f3a214"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:53:29.474533Z","signature_b64":"IiZWucI+hFMZYltcQzrgktLMRDj4za6ktazNQPv3nL4ZVJrxAYedA4NkyNOIFC3a6O9A6EOzUrykne2A7MfZAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6483e052688762eba7423970b519f6a6c1c357b95b85dae440e0f5a30bbc794a","last_reissued_at":"2026-07-05T11:53:29.474054Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:53:29.474054Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SABER: Switchable and Balanced Training for Efficient LLM Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jiaming Song, Kai Zhao, Lusheng Zhang, Qiang Zhang, Shien He, Tianjiao Li, Yanjun Zhao","submitted_at":"2025-08-08T11:27:48Z","abstract_excerpt":"Large language models (LLMs) empowered by chain-of-thought reasoning have achieved impressive accuracy on complex tasks but suffer from excessive inference costs and latency when applied uniformly to all problems. We propose SABER (Switchable and Balanced Training for Efficient LLM Reasoning), a reinforcement learning framework that endows LLMs with user-controllable, token-budgeted reasoning. SABER first profiles each training example's base-model thinking token usage and assigns it to one of the predefined budget tiers. During fine-tuning, the model is guided by system prompts and length-awa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.10026","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.10026/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.10026","created_at":"2026-07-05T11:53:29.474111+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.10026v1","created_at":"2026-07-05T11:53:29.474111+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.10026","created_at":"2026-07-05T11:53:29.474111+00:00"},{"alias_kind":"pith_short_12","alias_value":"MSB6AUTIQ5RO","created_at":"2026-07-05T11:53:29.474111+00:00"},{"alias_kind":"pith_short_16","alias_value":"MSB6AUTIQ5ROXJ2C","created_at":"2026-07-05T11:53:29.474111+00:00"},{"alias_kind":"pith_short_8","alias_value":"MSB6AUTI","created_at":"2026-07-05T11:53:29.474111+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17687","citing_title":"SuCo: Sufficiency-guided Continuous Adaptive Reasoning","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11625","citing_title":"Nice Fold or Hero Call: Learning Budget-Efficient Thinking for Adaptive Reasoning","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3","json":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3.json","graph_json":"https://pith.science/api/pith-number/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/graph.json","events_json":"https://pith.science/api/pith-number/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/events.json","paper":"https://pith.science/paper/MSB6AUTI"},"agent_actions":{"view_html":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3","download_json":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3.json","view_paper":"https://pith.science/paper/MSB6AUTI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.10026&json=true","fetch_graph":"https://pith.science/api/pith-number/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/graph.json","fetch_events":"https://pith.science/api/pith-number/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/action/storage_attestation","attest_author":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/action/author_attestation","sign_citation":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/action/citation_signature","submit_replication":"https://pith.science/pith/MSB6AUTIQ5ROXJ2CHFYLKGPWU3/action/replication_record"}},"created_at":"2026-07-05T11:53:29.474111+00:00","updated_at":"2026-07-05T11:53:29.474111+00:00"}