{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:SP7OOQMJFV7NWPGFDU5DCMWV5E","short_pith_number":"pith:SP7OOQMJ","schema_version":"1.0","canonical_sha256":"93fee741892d7edb3cc51d3a3132d5e90b0aa8772f162c315900938b5b1e2ae0","source":{"kind":"arxiv","id":"2210.05805","version":2},"attestation_state":"computed","paper":{"title":"Exploration via Elliptical Episodic Bonuses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mikael Henaff, Minqi Jiang, Roberta Raileanu, Tim Rockt\\\"aschel","submitted_at":"2022-10-11T22:10:23Z","abstract_excerpt":"In recent years, a number of reinforcement learning (RL) methods have been proposed to explore complex environments which differ across episodes. In this work, we show that the effectiveness of these methods critically relies on a count-based episodic term in their exploration bonus. As a result, despite their success in relatively simple, noise-free settings, these methods fall short in more realistic scenarios where the state space is vast and prone to noise. To address this limitation, we introduce Exploration via Elliptical Episodic Bonuses (E3B), a new method which extends count-based epi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.05805","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-10-11T22:10:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f5ac73d5aaeb7339705efe8c9436f7fc631cfa5455682c6bc10238a90922b9fb","abstract_canon_sha256":"c679f609fd7eca5ef1a274cd6d17c9bafee0fe3bf1680bc198d9bad5f78a63d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:30:44.384053Z","signature_b64":"0P/sE5rj9U0c6GQ4UwTV+wiHNru+deiayBSY/ttCHGbg1sd2Z7F+WbxajjaQJysBNCHYc+F8r9aVhWjIkkBQDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93fee741892d7edb3cc51d3a3132d5e90b0aa8772f162c315900938b5b1e2ae0","last_reissued_at":"2026-07-05T05:30:44.383512Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:30:44.383512Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploration via Elliptical Episodic Bonuses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mikael Henaff, Minqi Jiang, Roberta Raileanu, Tim Rockt\\\"aschel","submitted_at":"2022-10-11T22:10:23Z","abstract_excerpt":"In recent years, a number of reinforcement learning (RL) methods have been proposed to explore complex environments which differ across episodes. In this work, we show that the effectiveness of these methods critically relies on a count-based episodic term in their exploration bonus. As a result, despite their success in relatively simple, noise-free settings, these methods fall short in more realistic scenarios where the state space is vast and prone to noise. To address this limitation, we introduce Exploration via Elliptical Episodic Bonuses (E3B), a new method which extends count-based epi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.05805","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.05805/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.05805","created_at":"2026-07-05T05:30:44.383580+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.05805v2","created_at":"2026-07-05T05:30:44.383580+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.05805","created_at":"2026-07-05T05:30:44.383580+00:00"},{"alias_kind":"pith_short_12","alias_value":"SP7OOQMJFV7N","created_at":"2026-07-05T05:30:44.383580+00:00"},{"alias_kind":"pith_short_16","alias_value":"SP7OOQMJFV7NWPGF","created_at":"2026-07-05T05:30:44.383580+00:00"},{"alias_kind":"pith_short_8","alias_value":"SP7OOQMJ","created_at":"2026-07-05T05:30:44.383580+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":146,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E","json":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E.json","graph_json":"https://pith.science/api/pith-number/SP7OOQMJFV7NWPGFDU5DCMWV5E/graph.json","events_json":"https://pith.science/api/pith-number/SP7OOQMJFV7NWPGFDU5DCMWV5E/events.json","paper":"https://pith.science/paper/SP7OOQMJ"},"agent_actions":{"view_html":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E","download_json":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E.json","view_paper":"https://pith.science/paper/SP7OOQMJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.05805&json=true","fetch_graph":"https://pith.science/api/pith-number/SP7OOQMJFV7NWPGFDU5DCMWV5E/graph.json","fetch_events":"https://pith.science/api/pith-number/SP7OOQMJFV7NWPGFDU5DCMWV5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E/action/storage_attestation","attest_author":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E/action/author_attestation","sign_citation":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E/action/citation_signature","submit_replication":"https://pith.science/pith/SP7OOQMJFV7NWPGFDU5DCMWV5E/action/replication_record"}},"created_at":"2026-07-05T05:30:44.383580+00:00","updated_at":"2026-07-05T05:30:44.383580+00:00"}