{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SIWIAOUCZZOW3OA54MXUWT6EJK","short_pith_number":"pith:SIWIAOUC","schema_version":"1.0","canonical_sha256":"922c803a82ce5d6db81de32f4b4fc44abaaee509bc41d29da9e47903f53c8152","source":{"kind":"arxiv","id":"2105.01648","version":4},"attestation_state":"computed","paper":{"title":"On Lottery Tickets and Minimal Task Representations in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Henning Sprekeler, Marc Aurel Vischer, Robert Tjarko Lange","submitted_at":"2021-05-04T17:47:39Z","abstract_excerpt":"The lottery ticket hypothesis questions the role of overparameterization in supervised deep learning. But how is the performance of winning lottery tickets affected by the distributional shift inherent to reinforcement learning problems? In this work, we address this question by comparing sparse agents who have to address the non-stationarity of the exploration-exploitation problem with supervised agents trained to imitate an expert. We show that feed-forward networks trained with behavioural cloning compared to reinforcement learning can be pruned to higher levels of sparsity without performa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.01648","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-05-04T17:47:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ded7c368578ed0b2773fd16668c708384f7159f3f040d62e7e1ce546712f6fd4","abstract_canon_sha256":"3ccd474a459899e74c6c02ef96ed2dd3bf9e1b8f3336ea0a51b5f154dcdcc213"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:21:45.410474Z","signature_b64":"q0n9Q2JXLuv1sAVyBw8i809RCkK6LqB+ie/iJa34S3q76RLA/TXgz1J222OxMALDzhSGbhYeYSoSo2WMAPOpCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"922c803a82ce5d6db81de32f4b4fc44abaaee509bc41d29da9e47903f53c8152","last_reissued_at":"2026-07-05T04:21:45.410021Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:21:45.410021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Lottery Tickets and Minimal Task Representations in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Henning Sprekeler, Marc Aurel Vischer, Robert Tjarko Lange","submitted_at":"2021-05-04T17:47:39Z","abstract_excerpt":"The lottery ticket hypothesis questions the role of overparameterization in supervised deep learning. But how is the performance of winning lottery tickets affected by the distributional shift inherent to reinforcement learning problems? In this work, we address this question by comparing sparse agents who have to address the non-stationarity of the exploration-exploitation problem with supervised agents trained to imitate an expert. We show that feed-forward networks trained with behavioural cloning compared to reinforcement learning can be pruned to higher levels of sparsity without performa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.01648","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.01648/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.01648","created_at":"2026-07-05T04:21:45.410084+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.01648v4","created_at":"2026-07-05T04:21:45.410084+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.01648","created_at":"2026-07-05T04:21:45.410084+00:00"},{"alias_kind":"pith_short_12","alias_value":"SIWIAOUCZZOW","created_at":"2026-07-05T04:21:45.410084+00:00"},{"alias_kind":"pith_short_16","alias_value":"SIWIAOUCZZOW3OA5","created_at":"2026-07-05T04:21:45.410084+00:00"},{"alias_kind":"pith_short_8","alias_value":"SIWIAOUC","created_at":"2026-07-05T04:21:45.410084+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.17204","citing_title":"Network Sparsity Unlocks the Scaling Potential of Deep Reinforcement Learning","ref_index":60,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK","json":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK.json","graph_json":"https://pith.science/api/pith-number/SIWIAOUCZZOW3OA54MXUWT6EJK/graph.json","events_json":"https://pith.science/api/pith-number/SIWIAOUCZZOW3OA54MXUWT6EJK/events.json","paper":"https://pith.science/paper/SIWIAOUC"},"agent_actions":{"view_html":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK","download_json":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK.json","view_paper":"https://pith.science/paper/SIWIAOUC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.01648&json=true","fetch_graph":"https://pith.science/api/pith-number/SIWIAOUCZZOW3OA54MXUWT6EJK/graph.json","fetch_events":"https://pith.science/api/pith-number/SIWIAOUCZZOW3OA54MXUWT6EJK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK/action/storage_attestation","attest_author":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK/action/author_attestation","sign_citation":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK/action/citation_signature","submit_replication":"https://pith.science/pith/SIWIAOUCZZOW3OA54MXUWT6EJK/action/replication_record"}},"created_at":"2026-07-05T04:21:45.410084+00:00","updated_at":"2026-07-05T04:21:45.410084+00:00"}