{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:KVWDVK2L63CVANIAFZ3OPXRE7S","short_pith_number":"pith:KVWDVK2L","schema_version":"1.0","canonical_sha256":"556c3aab4bf6c55035002e76e7de24fc99d6a1b19a3105ecf4a3d5f8d30e341c","source":{"kind":"arxiv","id":"1706.10295","version":3},"attestation_state":"computed","paper":{"title":"Noisy Networks for Exploration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alex Graves, Bilal Piot, Charles Blundell, Demis Hassabis, Ian Osband, Jacob Menick, Meire Fortunato, Mohammad Gheshlaghi Azar, Olivier Pietquin, Remi Munos, Shane Legg, Vlad Mnih","submitted_at":"2017-06-30T17:56:19Z","abstract_excerpt":"We introduce NoisyNet, a deep reinforcement learning agent with parametric noise added to its weights, and show that the induced stochasticity of the agent's policy can be used to aid efficient exploration. The parameters of the noise are learned with gradient descent along with the remaining network weights. NoisyNet is straightforward to implement and adds little computational overhead. We find that replacing the conventional exploration heuristics for A3C, DQN and dueling agents (entropy reward and $\\epsilon$-greedy respectively) with NoisyNet yields substantially higher scores for a wide r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1706.10295","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2017-06-30T17:56:19Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"f599c580669695acf8cc3293671c65681f3e6a2c233475980b955ab299669cb8","abstract_canon_sha256":"6927ade1b43315b233063861a5fa00f5b951000e45bcd3c34c3d00c1d621f5ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:16:16.315527Z","signature_b64":"ZJs7rv7TG+DK62M6s9jK/P6wChW4TNJ1d1MrpFM5cNa3I6hPJPiwiHNRgu/yPBlzC7GM84wEbwX5s2ofR7h/BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"556c3aab4bf6c55035002e76e7de24fc99d6a1b19a3105ecf4a3d5f8d30e341c","last_reissued_at":"2026-07-05T00:16:16.315000Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:16:16.315000Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Noisy Networks for Exploration","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Alex Graves, Bilal Piot, Charles Blundell, Demis Hassabis, Ian Osband, Jacob Menick, Meire Fortunato, Mohammad Gheshlaghi Azar, Olivier Pietquin, Remi Munos, Shane Legg, Vlad Mnih","submitted_at":"2017-06-30T17:56:19Z","abstract_excerpt":"We introduce NoisyNet, a deep reinforcement learning agent with parametric noise added to its weights, and show that the induced stochasticity of the agent's policy can be used to aid efficient exploration. The parameters of the noise are learned with gradient descent along with the remaining network weights. NoisyNet is straightforward to implement and adds little computational overhead. We find that replacing the conventional exploration heuristics for A3C, DQN and dueling agents (entropy reward and $\\epsilon$-greedy respectively) with NoisyNet yields substantially higher scores for a wide r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1706.10295","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1706.10295/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1706.10295","created_at":"2026-07-05T00:16:16.315074+00:00"},{"alias_kind":"arxiv_version","alias_value":"1706.10295v3","created_at":"2026-07-05T00:16:16.315074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1706.10295","created_at":"2026-07-05T00:16:16.315074+00:00"},{"alias_kind":"pith_short_12","alias_value":"KVWDVK2L63CV","created_at":"2026-07-05T00:16:16.315074+00:00"},{"alias_kind":"pith_short_16","alias_value":"KVWDVK2L63CVANIA","created_at":"2026-07-05T00:16:16.315074+00:00"},{"alias_kind":"pith_short_8","alias_value":"KVWDVK2L","created_at":"2026-07-05T00:16:16.315074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03113","citing_title":"Experience-Driven Dynamic Exits for LLMs with Reinforcement Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"1906.08805","citing_title":"Finding Needles in a Moving Haystack: Prioritizing Alerts with Adversarial Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"1907.06077","citing_title":"Evolvability ES: Scalable and Direct Optimization of Evolvability","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2010.02193","citing_title":"Mastering Atari with Discrete World Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11697","citing_title":"Rainbow Deep Q-Learning with Kinematics-Aware Design for Cooperative Delta and 3-RRS Parallel Robot Insertion","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12667","citing_title":"Safe reinforcement learning with online filtering for fatigue-predictive human-robot task planning and allocation in production","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S","json":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S.json","graph_json":"https://pith.science/api/pith-number/KVWDVK2L63CVANIAFZ3OPXRE7S/graph.json","events_json":"https://pith.science/api/pith-number/KVWDVK2L63CVANIAFZ3OPXRE7S/events.json","paper":"https://pith.science/paper/KVWDVK2L"},"agent_actions":{"view_html":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S","download_json":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S.json","view_paper":"https://pith.science/paper/KVWDVK2L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1706.10295&json=true","fetch_graph":"https://pith.science/api/pith-number/KVWDVK2L63CVANIAFZ3OPXRE7S/graph.json","fetch_events":"https://pith.science/api/pith-number/KVWDVK2L63CVANIAFZ3OPXRE7S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S/action/storage_attestation","attest_author":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S/action/author_attestation","sign_citation":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S/action/citation_signature","submit_replication":"https://pith.science/pith/KVWDVK2L63CVANIAFZ3OPXRE7S/action/replication_record"}},"created_at":"2026-07-05T00:16:16.315074+00:00","updated_at":"2026-07-05T00:16:16.315074+00:00"}