{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SN4HBWKVII5QQ6LIYHBLVUYON3","short_pith_number":"pith:SN4HBWKV","schema_version":"1.0","canonical_sha256":"937870d955423b087968c1c2bad30e6eddf04b5bdc9cf1f85f7fbb353dd6f656","source":{"kind":"arxiv","id":"2506.08463","version":1},"attestation_state":"computed","paper":{"title":"How to Provably Improve Return Conditioned Supervised Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Dongruo Zhou, Pan Xu, Ruhan Wang, Yu Yang, Zhishuai Liu","submitted_at":"2025-06-10T05:37:51Z","abstract_excerpt":"In sequential decision-making problems, Return-Conditioned Supervised Learning (RCSL) has gained increasing recognition for its simplicity and stability in modern decision-making tasks. Unlike traditional offline reinforcement learning (RL) algorithms, RCSL frames policy learning as a supervised learning problem by taking both the state and return as input. This approach eliminates the instability often associated with temporal difference (TD) learning in offline RL. However, RCSL has been criticized for lacking the stitching property, meaning its performance is inherently limited by the quali"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.08463","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-10T05:37:51Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"7631a6e1b0c8084b76d975fc954d8717601ca345caaa7992caabd4965f420b49","abstract_canon_sha256":"7fb0b989dc8a0b4ca1914639b22e5a3d6ff1eec00342c82c37c9f7166f2dad96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:50.196699Z","signature_b64":"op4FxJFGLxVBCdslphs12y+nIRjOuv/tOdMu1GARA3lbwsclBaRCIYwQM+ZN2IM//OBiG0tsGtEqiYDlB/ibBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"937870d955423b087968c1c2bad30e6eddf04b5bdc9cf1f85f7fbb353dd6f656","last_reissued_at":"2026-07-05T11:18:50.196240Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:50.196240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How to Provably Improve Return Conditioned Supervised Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Dongruo Zhou, Pan Xu, Ruhan Wang, Yu Yang, Zhishuai Liu","submitted_at":"2025-06-10T05:37:51Z","abstract_excerpt":"In sequential decision-making problems, Return-Conditioned Supervised Learning (RCSL) has gained increasing recognition for its simplicity and stability in modern decision-making tasks. Unlike traditional offline reinforcement learning (RL) algorithms, RCSL frames policy learning as a supervised learning problem by taking both the state and return as input. This approach eliminates the instability often associated with temporal difference (TD) learning in offline RL. However, RCSL has been criticized for lacking the stitching property, meaning its performance is inherently limited by the quali"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.08463","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.08463/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.08463","created_at":"2026-07-05T11:18:50.196306+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.08463v1","created_at":"2026-07-05T11:18:50.196306+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.08463","created_at":"2026-07-05T11:18:50.196306+00:00"},{"alias_kind":"pith_short_12","alias_value":"SN4HBWKVII5Q","created_at":"2026-07-05T11:18:50.196306+00:00"},{"alias_kind":"pith_short_16","alias_value":"SN4HBWKVII5QQ6LI","created_at":"2026-07-05T11:18:50.196306+00:00"},{"alias_kind":"pith_short_8","alias_value":"SN4HBWKV","created_at":"2026-07-05T11:18:50.196306+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3","json":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3.json","graph_json":"https://pith.science/api/pith-number/SN4HBWKVII5QQ6LIYHBLVUYON3/graph.json","events_json":"https://pith.science/api/pith-number/SN4HBWKVII5QQ6LIYHBLVUYON3/events.json","paper":"https://pith.science/paper/SN4HBWKV"},"agent_actions":{"view_html":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3","download_json":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3.json","view_paper":"https://pith.science/paper/SN4HBWKV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.08463&json=true","fetch_graph":"https://pith.science/api/pith-number/SN4HBWKVII5QQ6LIYHBLVUYON3/graph.json","fetch_events":"https://pith.science/api/pith-number/SN4HBWKVII5QQ6LIYHBLVUYON3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3/action/storage_attestation","attest_author":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3/action/author_attestation","sign_citation":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3/action/citation_signature","submit_replication":"https://pith.science/pith/SN4HBWKVII5QQ6LIYHBLVUYON3/action/replication_record"}},"created_at":"2026-07-05T11:18:50.196306+00:00","updated_at":"2026-07-05T11:18:50.196306+00:00"}