{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:O5UYJHZOJRRWFXL7GROCISIU24","short_pith_number":"pith:O5UYJHZO","canonical_record":{"source":{"id":"2409.04840","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-07T14:38:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"63cbf319ea65ea33ef5838d90fb417834fcc672c1042e6f3653f127b1a4dcc0f","abstract_canon_sha256":"955b47191b9dbee63a739b84588eb2c99ea7526f3125778a6e72c8044096914b"},"schema_version":"1.0"},"canonical_sha256":"7769849f2e4c6362dd7f345c244914d732d1db1c86b1eb302c2732ca8e810c3d","source":{"kind":"arxiv","id":"2409.04840","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.04840","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"arxiv_version","alias_value":"2409.04840v2","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04840","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_12","alias_value":"O5UYJHZOJRRW","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_16","alias_value":"O5UYJHZOJRRWFXL7","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_8","alias_value":"O5UYJHZO","created_at":"2026-07-05T09:15:14Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:O5UYJHZOJRRWFXL7GROCISIU24","target":"record","payload":{"canonical_record":{"source":{"id":"2409.04840","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-07T14:38:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"63cbf319ea65ea33ef5838d90fb417834fcc672c1042e6f3653f127b1a4dcc0f","abstract_canon_sha256":"955b47191b9dbee63a739b84588eb2c99ea7526f3125778a6e72c8044096914b"},"schema_version":"1.0"},"canonical_sha256":"7769849f2e4c6362dd7f345c244914d732d1db1c86b1eb302c2732ca8e810c3d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:14.384350Z","signature_b64":"tAtPejqUjVk3/0nZB129falgQYWEWisfh23DnLEEPTux/MZfJ7TA+o8/Dwt+c3ahjd2wwcB3k2+5hcWERUb0DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7769849f2e4c6362dd7f345c244914d732d1db1c86b1eb302c2732ca8e810c3d","last_reissued_at":"2026-07-05T09:15:14.383886Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:14.383886Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2409.04840","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:15:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UoTuZC/KaaPdT53P+VXqOtQybG7Ln75Jwsnprac0GataAZQ8zHPNmpTvl8SJ3jEGkyPsw3u08FnnwjxeSEoVCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T15:54:22.083749Z"},"content_sha256":"18f80963a41434c843776ec73db3ae34fda46db66fbb9c4134eeff9f367de3ca","schema_version":"1.0","event_id":"sha256:18f80963a41434c843776ec73db3ae34fda46db66fbb9c4134eeff9f367de3ca"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:O5UYJHZOJRRWFXL7GROCISIU24","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Sample and Oracle Efficient Reinforcement Learning for MDPs with Linearly-Realizable Value Functions","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Zakaria Mhammedi","submitted_at":"2024-09-07T14:38:05Z","abstract_excerpt":"Designing sample-efficient and computationally feasible reinforcement learning (RL) algorithms is particularly challenging in environments with large or infinite state and action spaces. In this paper, we advance this effort by presenting an efficient algorithm for Markov Decision Processes (MDPs) where the state-action value function of any policy is linear in a given feature map. This challenging setting can model environments with infinite states and actions, strictly generalizes classic linear MDPs, and currently lacks a computationally efficient algorithm under online access to the MDP. S"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04840","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.04840/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:15:14Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CzYEi4youLahYP7pMDKusGNmSVZjQ1vGoSaLcqu/L87fjJ5mlNuTBtC4aS66WGmdKF9khFoRavyxn7ZY9jbKDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T15:54:22.084299Z"},"content_sha256":"f074739c45c4a0ca3097ac3aca01931587e13de85139598d1d1738f0d5bf0272","schema_version":"1.0","event_id":"sha256:f074739c45c4a0ca3097ac3aca01931587e13de85139598d1d1738f0d5bf0272"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/O5UYJHZOJRRWFXL7GROCISIU24/bundle.json","state_url":"https://pith.science/pith/O5UYJHZOJRRWFXL7GROCISIU24/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/O5UYJHZOJRRWFXL7GROCISIU24/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T15:54:22Z","links":{"resolver":"https://pith.science/pith/O5UYJHZOJRRWFXL7GROCISIU24","bundle":"https://pith.science/pith/O5UYJHZOJRRWFXL7GROCISIU24/bundle.json","state":"https://pith.science/pith/O5UYJHZOJRRWFXL7GROCISIU24/state.json","well_known_bundle":"https://pith.science/.well-known/pith/O5UYJHZOJRRWFXL7GROCISIU24/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:O5UYJHZOJRRWFXL7GROCISIU24","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"955b47191b9dbee63a739b84588eb2c99ea7526f3125778a6e72c8044096914b","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-07T14:38:05Z","title_canon_sha256":"63cbf319ea65ea33ef5838d90fb417834fcc672c1042e6f3653f127b1a4dcc0f"},"schema_version":"1.0","source":{"id":"2409.04840","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.04840","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"arxiv_version","alias_value":"2409.04840v2","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04840","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_12","alias_value":"O5UYJHZOJRRW","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_16","alias_value":"O5UYJHZOJRRWFXL7","created_at":"2026-07-05T09:15:14Z"},{"alias_kind":"pith_short_8","alias_value":"O5UYJHZO","created_at":"2026-07-05T09:15:14Z"}],"graph_snapshots":[{"event_id":"sha256:f074739c45c4a0ca3097ac3aca01931587e13de85139598d1d1738f0d5bf0272","target":"graph","created_at":"2026-07-05T09:15:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.04840/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Designing sample-efficient and computationally feasible reinforcement learning (RL) algorithms is particularly challenging in environments with large or infinite state and action spaces. In this paper, we advance this effort by presenting an efficient algorithm for Markov Decision Processes (MDPs) where the state-action value function of any policy is linear in a given feature map. This challenging setting can model environments with infinite states and actions, strictly generalizes classic linear MDPs, and currently lacks a computationally efficient algorithm under online access to the MDP. S","authors_text":"Zakaria Mhammedi","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-07T14:38:05Z","title":"Sample and Oracle Efficient Reinforcement Learning for MDPs with Linearly-Realizable Value Functions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04840","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:18f80963a41434c843776ec73db3ae34fda46db66fbb9c4134eeff9f367de3ca","target":"record","created_at":"2026-07-05T09:15:14Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"955b47191b9dbee63a739b84588eb2c99ea7526f3125778a6e72c8044096914b","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-07T14:38:05Z","title_canon_sha256":"63cbf319ea65ea33ef5838d90fb417834fcc672c1042e6f3653f127b1a4dcc0f"},"schema_version":"1.0","source":{"id":"2409.04840","kind":"arxiv","version":2}},"canonical_sha256":"7769849f2e4c6362dd7f345c244914d732d1db1c86b1eb302c2732ca8e810c3d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7769849f2e4c6362dd7f345c244914d732d1db1c86b1eb302c2732ca8e810c3d","first_computed_at":"2026-07-05T09:15:14.383886Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:15:14.383886Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tAtPejqUjVk3/0nZB129falgQYWEWisfh23DnLEEPTux/MZfJ7TA+o8/Dwt+c3ahjd2wwcB3k2+5hcWERUb0DA==","signature_status":"signed_v1","signed_at":"2026-07-05T09:15:14.384350Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.04840","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:18f80963a41434c843776ec73db3ae34fda46db66fbb9c4134eeff9f367de3ca","sha256:f074739c45c4a0ca3097ac3aca01931587e13de85139598d1d1738f0d5bf0272"],"state_sha256":"990e814b0959362a59973fe59d707cff0448f395aabe343f13acbe3da98d5ae3"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"CuRNX3W8PwQKsd+8DtuMxID/RQ0ivWYpkBkMSC+jXR7ki2vbvoeIMK0XzEnszOUlt61rAeROiqIaZftXk/QhBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T15:54:22.088596Z","bundle_sha256":"4a10f52f51247c9aec1a5a8b7208de74c58a2d475437c259af141c1cf4c72abd"}}