{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VRPAT3OXV3VYSB5X4L5UUTD7WI","short_pith_number":"pith:VRPAT3OX","schema_version":"1.0","canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","source":{"kind":"arxiv","id":"2311.02215","version":1},"attestation_state":"computed","paper":{"title":"Towards model-free RL algorithms that scale well with unstructured data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Joseph Modayil, Zaheer Abbas","submitted_at":"2023-11-03T20:03:54Z","abstract_excerpt":"Conventional reinforcement learning (RL) algorithms exhibit broad generality in their theoretical formulation and high performance on several challenging domains when combined with powerful function approximation. However, developing RL algorithms that perform well across problems with unstructured observations at scale remains challenging because most function approximation methods rely on externally provisioned knowledge about the structure of the input for good performance (e.g. convolutional networks, graph neural networks, tile-coding). A common practice in RL is to evaluate algorithms on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.02215","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T20:03:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e765ea6add29c731a75e033913534a6074ec96940b952223a08c7b8a5dedf0a6","abstract_canon_sha256":"22abb74612f5aaeb6e8c403f4fc86d9c060788c64c2312d0617ca20b6ab0a094"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:07.594443Z","signature_b64":"Ob3Pp7+oGWQW07xGM/nLScjq9rLB3TaXiwLWekTbDCIFDJaGfrWGuC1M63iYsBtDCSa//Hqg5nkcEoMHhHaJCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ac5e09edd7aeeb8907b7e2fb4a4c7fb22c60b9327547d182f419c578d1254ae8","last_reissued_at":"2026-07-05T07:09:07.594041Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:07.594041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards model-free RL algorithms that scale well with unstructured data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Joseph Modayil, Zaheer Abbas","submitted_at":"2023-11-03T20:03:54Z","abstract_excerpt":"Conventional reinforcement learning (RL) algorithms exhibit broad generality in their theoretical formulation and high performance on several challenging domains when combined with powerful function approximation. However, developing RL algorithms that perform well across problems with unstructured observations at scale remains challenging because most function approximation methods rely on externally provisioned knowledge about the structure of the input for good performance (e.g. convolutional networks, graph neural networks, tile-coding). A common practice in RL is to evaluate algorithms on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02215","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.02215","created_at":"2026-07-05T07:09:07.594091+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.02215v1","created_at":"2026-07-05T07:09:07.594091+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02215","created_at":"2026-07-05T07:09:07.594091+00:00"},{"alias_kind":"pith_short_12","alias_value":"VRPAT3OXV3VY","created_at":"2026-07-05T07:09:07.594091+00:00"},{"alias_kind":"pith_short_16","alias_value":"VRPAT3OXV3VYSB5X","created_at":"2026-07-05T07:09:07.594091+00:00"},{"alias_kind":"pith_short_8","alias_value":"VRPAT3OX","created_at":"2026-07-05T07:09:07.594091+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.09523","citing_title":"An Analysis of Action-Value Temporal-Difference Methods That Learn State Values","ref_index":12,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI","json":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI.json","graph_json":"https://pith.science/api/pith-number/VRPAT3OXV3VYSB5X4L5UUTD7WI/graph.json","events_json":"https://pith.science/api/pith-number/VRPAT3OXV3VYSB5X4L5UUTD7WI/events.json","paper":"https://pith.science/paper/VRPAT3OX"},"agent_actions":{"view_html":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI","download_json":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI.json","view_paper":"https://pith.science/paper/VRPAT3OX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.02215&json=true","fetch_graph":"https://pith.science/api/pith-number/VRPAT3OXV3VYSB5X4L5UUTD7WI/graph.json","fetch_events":"https://pith.science/api/pith-number/VRPAT3OXV3VYSB5X4L5UUTD7WI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/action/storage_attestation","attest_author":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/action/author_attestation","sign_citation":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/action/citation_signature","submit_replication":"https://pith.science/pith/VRPAT3OXV3VYSB5X4L5UUTD7WI/action/replication_record"}},"created_at":"2026-07-05T07:09:07.594091+00:00","updated_at":"2026-07-05T07:09:07.594091+00:00"}