{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6OZIBYEQAR2EGZH3PBHO5AMC5T","short_pith_number":"pith:6OZIBYEQ","schema_version":"1.0","canonical_sha256":"f3b280e09004744364fb784eee8182ecf2fa31736ad4d7d372dc06c4100ac4d7","source":{"kind":"arxiv","id":"2502.08632","version":1},"attestation_state":"computed","paper":{"title":"Necessary and Sufficient Oracles: Toward a Computational Taxonomy For Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CC"],"primary_cat":"cs.LG","authors_text":"Dhruv Rohatgi, Dylan J. Foster","submitted_at":"2025-02-12T18:47:13Z","abstract_excerpt":"Algorithms for reinforcement learning (RL) in large state spaces crucially rely on supervised learning subroutines to estimate objects such as value functions or transition probabilities. Since only the simplest supervised learning problems can be solved provably and efficiently, practical performance of an RL algorithm depends on which of these supervised learning \"oracles\" it assumes access to (and how they are implemented). But which oracles are better or worse? Is there a minimal oracle?\n  In this work, we clarify the impact of the choice of supervised learning oracle on the computational "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08632","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-12T18:47:13Z","cross_cats_sorted":["cs.CC"],"title_canon_sha256":"7f6387c38e9bfed5489c33e9b750b7d35b0a9a19aa6362550e4f168b36c39f93","abstract_canon_sha256":"4a1fc6e222a3c0fa733622f1accd692cb61150b5ef628b8d99ef0ba79c5856a7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:25.684194Z","signature_b64":"bAL7AIENmvH1UUAg1RKZVuj5kvb2BEi65eeSBSteP5ydZUdwRJImLgZd7OyEcPFmh5HMsHpXa9lQ4Jm6em5JBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f3b280e09004744364fb784eee8182ecf2fa31736ad4d7d372dc06c4100ac4d7","last_reissued_at":"2026-07-05T10:13:25.683696Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:25.683696Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Necessary and Sufficient Oracles: Toward a Computational Taxonomy For Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CC"],"primary_cat":"cs.LG","authors_text":"Dhruv Rohatgi, Dylan J. Foster","submitted_at":"2025-02-12T18:47:13Z","abstract_excerpt":"Algorithms for reinforcement learning (RL) in large state spaces crucially rely on supervised learning subroutines to estimate objects such as value functions or transition probabilities. Since only the simplest supervised learning problems can be solved provably and efficiently, practical performance of an RL algorithm depends on which of these supervised learning \"oracles\" it assumes access to (and how they are implemented). But which oracles are better or worse? Is there a minimal oracle?\n  In this work, we clarify the impact of the choice of supervised learning oracle on the computational "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08632","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08632/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08632","created_at":"2026-07-05T10:13:25.683756+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08632v1","created_at":"2026-07-05T10:13:25.683756+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08632","created_at":"2026-07-05T10:13:25.683756+00:00"},{"alias_kind":"pith_short_12","alias_value":"6OZIBYEQAR2E","created_at":"2026-07-05T10:13:25.683756+00:00"},{"alias_kind":"pith_short_16","alias_value":"6OZIBYEQAR2EGZH3","created_at":"2026-07-05T10:13:25.683756+00:00"},{"alias_kind":"pith_short_8","alias_value":"6OZIBYEQ","created_at":"2026-07-05T10:13:25.683756+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.13934","citing_title":"Why Code, Why Now: An Information-Theoretic Perspective on the Limits of Machine Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01242","citing_title":"Breaking the Computational Barrier: Provably Efficient Actor-Critic for Low-Rank MDPs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04855","citing_title":"The Role of Generator Access in Autoregressive Post-Training","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T","json":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T.json","graph_json":"https://pith.science/api/pith-number/6OZIBYEQAR2EGZH3PBHO5AMC5T/graph.json","events_json":"https://pith.science/api/pith-number/6OZIBYEQAR2EGZH3PBHO5AMC5T/events.json","paper":"https://pith.science/paper/6OZIBYEQ"},"agent_actions":{"view_html":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T","download_json":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T.json","view_paper":"https://pith.science/paper/6OZIBYEQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08632&json=true","fetch_graph":"https://pith.science/api/pith-number/6OZIBYEQAR2EGZH3PBHO5AMC5T/graph.json","fetch_events":"https://pith.science/api/pith-number/6OZIBYEQAR2EGZH3PBHO5AMC5T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T/action/storage_attestation","attest_author":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T/action/author_attestation","sign_citation":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T/action/citation_signature","submit_replication":"https://pith.science/pith/6OZIBYEQAR2EGZH3PBHO5AMC5T/action/replication_record"}},"created_at":"2026-07-05T10:13:25.683756+00:00","updated_at":"2026-07-05T10:13:25.683756+00:00"}