{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:VRPAT3OXV3VYSB5X4L5UUTD7WI","short_pith_number":"pith:VRPAT3OX","canonical_record":{"source":{"id":"2311.02215","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T20:03:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e765ea6add29c731a75e033913534a6074ec96940b952223a08c7b8a5dedf0a6","abstract_canon_sha256":"22abb74612f5aaeb6e8c403f4fc86d9c060788c64c2312d0617ca20b6ab0a094"},"schema_version":"1.0"},"canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","source":{"kind":"arxiv","id":"2311.02215","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2311.02215","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"arxiv_version","alias_value":"2311.02215v1","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02215","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"pith_short_12","alias_value":"VRPAT3OXV3VY","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"pith_short_16","alias_value":"VRPAT3OXV3VYSB5X","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"pith_short_8","alias_value":"VRPAT3OX","created_at":"2026-07-05T07:09:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:VRPAT3OXV3VYSB5X4L5UUTD7WI","target":"record","payload":{"canonical_record":{"source":{"id":"2311.02215","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T20:03:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e765ea6add29c731a75e033913534a6074ec96940b952223a08c7b8a5dedf0a6","abstract_canon_sha256":"22abb74612f5aaeb6e8c403f4fc86d9c060788c64c2312d0617ca20b6ab0a094"},"schema_version":"1.0"},"canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:07.594443Z","signature_b64":"Ob3Pp7+oGWQW07xGM/nLScjq9rLB3TaXiwLWekTbDCIFDJaGfrWGuC1M63iYsBtDCSa//Hqg5nkcEoMHhHaJCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","last_reissued_at":"2026-07-05T07:09:07.594041Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:07.594041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2311.02215","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:09:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bKx8lWrh0Ji6LpdNZCpiW8UjDgKUZoDqaSBI/5d2oA9TacnGRmU9fvGjZ9ktxIQOaQoSiPqf9K81kiyLFPEeDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T10:25:17.359834Z"},"content_sha256":"ccd0093674a5ac00ce20475ce5b0f6c154be83fa262227a0c055d14f8c4645f9","schema_version":"1.0","event_id":"sha256:ccd0093674a5ac00ce20475ce5b0f6c154be83fa262227a0c055d14f8c4645f9"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:VRPAT3OXV3VYSB5X4L5UUTD7WI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Towards model-free RL algorithms that scale well with unstructured data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Joseph Modayil, Zaheer Abbas","submitted_at":"2023-11-03T20:03:54Z","abstract_excerpt":"Conventional reinforcement learning (RL) algorithms exhibit broad generality in their theoretical formulation and high performance on several challenging domains when combined with powerful function approximation. However, developing RL algorithms that perform well across problems with unstructured observations at scale remains challenging because most function approximation methods rely on externally provisioned knowledge about the structure of the input for good performance (e.g. convolutional networks, graph neural networks, tile-coding). A common practice in RL is to evaluate algorithms on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02215","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:09:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kldPjAFU8oUoQgIpKaPnqEtm8lrslEv/YMJwZC3PW+HO9rgWNxxKkRine3Htz+YC4Gpd2X7u+KGYMNrzprPGDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T10:25:17.360895Z"},"content_sha256":"ab4625ff220e078843e13f9326d4ab854316f1909b902c673271f99ec86952a7","schema_version":"1.0","event_id":"sha256:ab4625ff220e078843e13f9326d4ab854316f1909b902c673271f99ec86952a7"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/bundle.json","state_url":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T10:25:17Z","links":{"resolver":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI","bundle":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/bundle.json","state":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:VRPAT3OXV3VYSB5X4L5UUTD7WI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"22abb74612f5aaeb6e8c403f4fc86d9c060788c64c2312d0617ca20b6ab0a094","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T20:03:54Z","title_canon_sha256":"e765ea6add29c731a75e033913534a6074ec96940b952223a08c7b8a5dedf0a6"},"schema_version":"1.0","source":{"id":"2311.02215","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2311.02215","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"arxiv_version","alias_value":"2311.02215v1","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02215","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"pith_short_12","alias_value":"VRPAT3OXV3VY","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"pith_short_16","alias_value":"VRPAT3OXV3VYSB5X","created_at":"2026-07-05T07:09:07Z"},{"alias_kind":"pith_short_8","alias_value":"VRPAT3OX","created_at":"2026-07-05T07:09:07Z"}],"graph_snapshots":[{"event_id":"sha256:ab4625ff220e078843e13f9326d4ab854316f1909b902c673271f99ec86952a7","target":"graph","created_at":"2026-07-05T07:09:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2311.02215/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Conventional reinforcement learning (RL) algorithms exhibit broad generality in their theoretical formulation and high performance on several challenging domains when combined with powerful function approximation. However, developing RL algorithms that perform well across problems with unstructured observations at scale remains challenging because most function approximation methods rely on externally provisioned knowledge about the structure of the input for good performance (e.g. convolutional networks, graph neural networks, tile-coding). A common practice in RL is to evaluate algorithms on","authors_text":"Joseph Modayil, Zaheer Abbas","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T20:03:54Z","title":"Towards model-free RL algorithms that scale well with unstructured data"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02215","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ccd0093674a5ac00ce20475ce5b0f6c154be83fa262227a0c055d14f8c4645f9","target":"record","created_at":"2026-07-05T07:09:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"22abb74612f5aaeb6e8c403f4fc86d9c060788c64c2312d0617ca20b6ab0a094","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T20:03:54Z","title_canon_sha256":"e765ea6add29c731a75e033913534a6074ec96940b952223a08c7b8a5dedf0a6"},"schema_version":"1.0","source":{"id":"2311.02215","kind":"arxiv","version":1}},"canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","first_computed_at":"2026-07-05T07:09:07.594041Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:09:07.594041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Ob3Pp7+oGWQW07xGM/nLScjq9rLB3TaXiwLWekTbDCIFDJaGfrWGuC1M63iYsBtDCSa//Hqg5nkcEoMHhHaJCA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:09:07.594443Z","signed_message":"canonical_sha256_bytes"},"source_id":"2311.02215","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ccd0093674a5ac00ce20475ce5b0f6c154be83fa262227a0c055d14f8c4645f9","sha256:ab4625ff220e078843e13f9326d4ab854316f1909b902c673271f99ec86952a7"],"state_sha256":"3f133a0b6eea455d915098441cd1231209fd55dc79060b8184ea2e54a58e95fd"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"BW9+eqE2TQ5l35yEvm6fHM1ZIm8j6x7jWq7inllI6hUFI+HPMmy5zQijZ3tQiVkndA1Yg49vBS6To4shBKBwCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T10:25:17.368224Z","bundle_sha256":"9847015558d0ced96b71818cf88bff32e17be9d5389a9309d32d845fe56b596c"}}