{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XWTECPAJ3L6TVL2WZPXPU2IJWQ","short_pith_number":"pith:XWTECPAJ","schema_version":"1.0","canonical_sha256":"bda6413c09dafd3aaf56cbeefa6909b40cdce5803571a8b02966401a20386ecf","source":{"kind":"arxiv","id":"2505.22866","version":1},"attestation_state":"computed","paper":{"title":"Scaling Offline RL via Efficient and Expressive Shortcut Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bradley Guo, Gokul Swamy, Kiante Brantley, Nicolas Espinosa-Dice, Owen Oertell, Wen Sun, Yiding Chen, Yiyi Zhang","submitted_at":"2025-05-28T20:59:22Z","abstract_excerpt":"Diffusion and flow models have emerged as powerful generative approaches capable of modeling diverse and multimodal behavior. However, applying these models to offline reinforcement learning (RL) remains challenging due to the iterative nature of their noise sampling processes, making policy optimization difficult. In this paper, we introduce Scalable Offline Reinforcement Learning (SORL), a new offline RL algorithm that leverages shortcut models - a novel class of generative models - to scale both training and inference. SORL's policy can capture complex data distributions and can be trained "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.22866","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-28T20:59:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e31a629278a966a2b60d2e4c3aa3f35b774d53ba2766516dd5c4b6966ba5946e","abstract_canon_sha256":"44aa202c570fdf5b08ce23f7aca96b52c967c39507bb6c42c68e29378174487e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:51.103535Z","signature_b64":"wtX7iPI19oN98Vw73hYhZeOQtaJm7iIqj5pwHXTNylfVoQhT8JkUpN7AB9haNUTcwlYKxsntHW6zpC1aq9yKAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bda6413c09dafd3aaf56cbeefa6909b40cdce5803571a8b02966401a20386ecf","last_reissued_at":"2026-07-05T11:11:51.102811Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:51.102811Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Offline RL via Efficient and Expressive Shortcut Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bradley Guo, Gokul Swamy, Kiante Brantley, Nicolas Espinosa-Dice, Owen Oertell, Wen Sun, Yiding Chen, Yiyi Zhang","submitted_at":"2025-05-28T20:59:22Z","abstract_excerpt":"Diffusion and flow models have emerged as powerful generative approaches capable of modeling diverse and multimodal behavior. However, applying these models to offline reinforcement learning (RL) remains challenging due to the iterative nature of their noise sampling processes, making policy optimization difficult. In this paper, we introduce Scalable Offline Reinforcement Learning (SORL), a new offline RL algorithm that leverages shortcut models - a novel class of generative models - to scale both training and inference. SORL's policy can capture complex data distributions and can be trained "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.22866","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.22866/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.22866","created_at":"2026-07-05T11:11:51.102878+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.22866v1","created_at":"2026-07-05T11:11:51.102878+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.22866","created_at":"2026-07-05T11:11:51.102878+00:00"},{"alias_kind":"pith_short_12","alias_value":"XWTECPAJ3L6T","created_at":"2026-07-05T11:11:51.102878+00:00"},{"alias_kind":"pith_short_16","alias_value":"XWTECPAJ3L6TVL2W","created_at":"2026-07-05T11:11:51.102878+00:00"},{"alias_kind":"pith_short_8","alias_value":"XWTECPAJ","created_at":"2026-07-05T11:11:51.102878+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11087","citing_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04333","citing_title":"What Does Flow Matching Bring To TD Learning?","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22229","citing_title":"Preserve Support, Not Correspondence: Dynamic Routing for Offline Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01356","citing_title":"Model-Based Proactive Cost Generation for Learning Safe Policies Offline with Limited Violation Data","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01663","citing_title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07727","citing_title":"Drifting Field Policy: A One-Step Generative Policy via Wasserstein Gradient Flow","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ","json":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ.json","graph_json":"https://pith.science/api/pith-number/XWTECPAJ3L6TVL2WZPXPU2IJWQ/graph.json","events_json":"https://pith.science/api/pith-number/XWTECPAJ3L6TVL2WZPXPU2IJWQ/events.json","paper":"https://pith.science/paper/XWTECPAJ"},"agent_actions":{"view_html":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ","download_json":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ.json","view_paper":"https://pith.science/paper/XWTECPAJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.22866&json=true","fetch_graph":"https://pith.science/api/pith-number/XWTECPAJ3L6TVL2WZPXPU2IJWQ/graph.json","fetch_events":"https://pith.science/api/pith-number/XWTECPAJ3L6TVL2WZPXPU2IJWQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ/action/storage_attestation","attest_author":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ/action/author_attestation","sign_citation":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ/action/citation_signature","submit_replication":"https://pith.science/pith/XWTECPAJ3L6TVL2WZPXPU2IJWQ/action/replication_record"}},"created_at":"2026-07-05T11:11:51.102878+00:00","updated_at":"2026-07-05T11:11:51.102878+00:00"}