{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:222KYOCWFFMWHGQ3ISQQK2GNSY","short_pith_number":"pith:222KYOCW","schema_version":"1.0","canonical_sha256":"d6b4ac38562959639a1b44a10568cd962930eafce58c642c7267c59b373c67a8","source":{"kind":"arxiv","id":"2504.07891","version":2},"attestation_state":"computed","paper":{"title":"SpecReason: Fast and Accurate Inference-Time Compute via Speculative Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gabriele Oliaro, Ravi Netravali, Rui Pan, Yinwei Dai, Zhihao Jia, Zhihao Zhang","submitted_at":"2025-04-10T16:05:19Z","abstract_excerpt":"Recent advances in inference-time compute have significantly improved performance on complex tasks by generating long chains of thought (CoTs) using Large Reasoning Models (LRMs). However, this improved accuracy comes at the cost of high inference latency due to the length of generated reasoning sequences and the autoregressive nature of decoding. Our key insight in tackling these overheads is that LRM inference, and the reasoning that it embeds, is highly tolerant of approximations: complex tasks are typically broken down into simpler steps, each of which brings utility based on the semantic "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07891","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-10T16:05:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"6c5f5d71968da75691f81501ec73e5d6f0aaa35183bcba3e9d710b91e8133647","abstract_canon_sha256":"1011f79c7d4099bd79ba90419d779c7220c0e0509199af029d726ad4c153be2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:21.537494Z","signature_b64":"h9iDFCe6WiDcHCl+bKjGfS1ujJreZEujBLZf5Sw3kvqEpPTE0osdTWxmb9PC1InhtRJeJPvvlnr896dXKczBAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d6b4ac38562959639a1b44a10568cd962930eafce58c642c7267c59b373c67a8","last_reissued_at":"2026-07-05T11:04:21.536950Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:21.536950Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SpecReason: Fast and Accurate Inference-Time Compute via Speculative Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gabriele Oliaro, Ravi Netravali, Rui Pan, Yinwei Dai, Zhihao Jia, Zhihao Zhang","submitted_at":"2025-04-10T16:05:19Z","abstract_excerpt":"Recent advances in inference-time compute have significantly improved performance on complex tasks by generating long chains of thought (CoTs) using Large Reasoning Models (LRMs). However, this improved accuracy comes at the cost of high inference latency due to the length of generated reasoning sequences and the autoregressive nature of decoding. Our key insight in tackling these overheads is that LRM inference, and the reasoning that it embeds, is highly tolerant of approximations: complex tasks are typically broken down into simpler steps, each of which brings utility based on the semantic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07891","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07891/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07891","created_at":"2026-07-05T11:04:21.537023+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07891v2","created_at":"2026-07-05T11:04:21.537023+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07891","created_at":"2026-07-05T11:04:21.537023+00:00"},{"alias_kind":"pith_short_12","alias_value":"222KYOCWFFMW","created_at":"2026-07-05T11:04:21.537023+00:00"},{"alias_kind":"pith_short_16","alias_value":"222KYOCWFFMWHGQ3","created_at":"2026-07-05T11:04:21.537023+00:00"},{"alias_kind":"pith_short_8","alias_value":"222KYOCW","created_at":"2026-07-05T11:04:21.537023+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19755","citing_title":"SafeSpec: Fast and Safe LLM via Dynamic Reflective Sampling","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26940","citing_title":"Select to Think: Unlocking SLM Potential with Local Sufficiency","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.05110","citing_title":"GlimpRouter: Efficient Collaborative Inference by Glimpsing One Token of Thoughts","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10195","citing_title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16419","citing_title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26940","citing_title":"Select to Think: Unlocking SLM Potential with Local Sufficiency","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08580","citing_title":"Slipstream: Trajectory-Grounded Compaction Validation for Long-Horizon Agents","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10195","citing_title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06165","citing_title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","ref_index":197,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06116","citing_title":"Policy-Guided Stepwise Model Routing for Cost-Effective Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14847","citing_title":"TrigReason: Trigger-Based Collaboration between Small and Large Reasoning Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15244","citing_title":"From Tokens to Steps: Verification-Aware Speculative Decoding for Efficient Multi-Step Reasoning","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY","json":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY.json","graph_json":"https://pith.science/api/pith-number/222KYOCWFFMWHGQ3ISQQK2GNSY/graph.json","events_json":"https://pith.science/api/pith-number/222KYOCWFFMWHGQ3ISQQK2GNSY/events.json","paper":"https://pith.science/paper/222KYOCW"},"agent_actions":{"view_html":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY","download_json":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY.json","view_paper":"https://pith.science/paper/222KYOCW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07891&json=true","fetch_graph":"https://pith.science/api/pith-number/222KYOCWFFMWHGQ3ISQQK2GNSY/graph.json","fetch_events":"https://pith.science/api/pith-number/222KYOCWFFMWHGQ3ISQQK2GNSY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY/action/storage_attestation","attest_author":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY/action/author_attestation","sign_citation":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY/action/citation_signature","submit_replication":"https://pith.science/pith/222KYOCWFFMWHGQ3ISQQK2GNSY/action/replication_record"}},"created_at":"2026-07-05T11:04:21.537023+00:00","updated_at":"2026-07-05T11:04:21.537023+00:00"}