{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SUKGKFOJPUJKOG377QZTKAU7GE","short_pith_number":"pith:SUKGKFOJ","schema_version":"1.0","canonical_sha256":"95146515c97d12a71b7ffc3335029f3121cc21a81cf666f453af6ce6b908321e","source":{"kind":"arxiv","id":"2505.20268","version":2},"attestation_state":"computed","paper":{"title":"Outcome-Based Online Reinforcement Learning: Algorithms and Fundamental Limits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Alexander Rakhlin, Fan Chen, Tengyang Xie, Zeyu Jia","submitted_at":"2025-05-26T17:44:08Z","abstract_excerpt":"Reinforcement learning with outcome-based feedback faces a fundamental challenge: when rewards are only observed at trajectory endpoints, how do we assign credit to the right actions? This paper provides the first comprehensive analysis of this problem in online RL with general function approximation. We develop a provably sample-efficient algorithm achieving $\\widetilde{O}({C_{\\rm cov} H^3}/{\\epsilon^2})$ sample complexity, where $C_{\\rm cov}$ is the coverability coefficient of the underlying MDP. By leveraging general function approximation, our approach works effectively in large or infinit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.20268","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-26T17:44:08Z","cross_cats_sorted":["cs.AI","math.ST","stat.ML","stat.TH"],"title_canon_sha256":"9d0de54fca53262ad0e8443c4f627f59bf4445c95dbccfa7f4a10ad7890a2f02","abstract_canon_sha256":"e1b54ea20bdc17782476f28776569ef10c84bc0405204cffe46261f6b41b7611"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:29.796173Z","signature_b64":"VFUV+Ngl9MkGPCFv5XwNrt3ZQf7VxXjp7z6XBwWWDfVgPkCWehp9i1/LGs0HEC/jwvauP+5VFGkqp5kAz1rKCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95146515c97d12a71b7ffc3335029f3121cc21a81cf666f453af6ce6b908321e","last_reissued_at":"2026-07-05T11:42:29.795657Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:29.795657Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Outcome-Based Online Reinforcement Learning: Algorithms and Fundamental Limits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Alexander Rakhlin, Fan Chen, Tengyang Xie, Zeyu Jia","submitted_at":"2025-05-26T17:44:08Z","abstract_excerpt":"Reinforcement learning with outcome-based feedback faces a fundamental challenge: when rewards are only observed at trajectory endpoints, how do we assign credit to the right actions? This paper provides the first comprehensive analysis of this problem in online RL with general function approximation. We develop a provably sample-efficient algorithm achieving $\\widetilde{O}({C_{\\rm cov} H^3}/{\\epsilon^2})$ sample complexity, where $C_{\\rm cov}$ is the coverability coefficient of the underlying MDP. By leveraging general function approximation, our approach works effectively in large or infinit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20268","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20268/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.20268","created_at":"2026-07-05T11:42:29.795719+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.20268v2","created_at":"2026-07-05T11:42:29.795719+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20268","created_at":"2026-07-05T11:42:29.795719+00:00"},{"alias_kind":"pith_short_12","alias_value":"SUKGKFOJPUJK","created_at":"2026-07-05T11:42:29.795719+00:00"},{"alias_kind":"pith_short_16","alias_value":"SUKGKFOJPUJKOG37","created_at":"2026-07-05T11:42:29.795719+00:00"},{"alias_kind":"pith_short_8","alias_value":"SUKGKFOJ","created_at":"2026-07-05T11:42:29.795719+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18531","citing_title":"When Does Trajectory-Level Supervision Permit Efficient Offline Reinforcement Learning?","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08539","citing_title":"On the optimization dynamics of RLVR: Gradient gap and step size thresholds","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07049","citing_title":"Towards Differentially Private Reinforcement Learning with General Function Approximation","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE","json":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE.json","graph_json":"https://pith.science/api/pith-number/SUKGKFOJPUJKOG377QZTKAU7GE/graph.json","events_json":"https://pith.science/api/pith-number/SUKGKFOJPUJKOG377QZTKAU7GE/events.json","paper":"https://pith.science/paper/SUKGKFOJ"},"agent_actions":{"view_html":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE","download_json":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE.json","view_paper":"https://pith.science/paper/SUKGKFOJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.20268&json=true","fetch_graph":"https://pith.science/api/pith-number/SUKGKFOJPUJKOG377QZTKAU7GE/graph.json","fetch_events":"https://pith.science/api/pith-number/SUKGKFOJPUJKOG377QZTKAU7GE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE/action/storage_attestation","attest_author":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE/action/author_attestation","sign_citation":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE/action/citation_signature","submit_replication":"https://pith.science/pith/SUKGKFOJPUJKOG377QZTKAU7GE/action/replication_record"}},"created_at":"2026-07-05T11:42:29.795719+00:00","updated_at":"2026-07-05T11:42:29.795719+00:00"}