{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ABEQFEI24OGMTR4QS6CCGG4FJS","short_pith_number":"pith:ABEQFEI2","schema_version":"1.0","canonical_sha256":"004902911ae38cc9c7909784231b854c8ef0d32a3a8b4908a49410d50633c668","source":{"kind":"arxiv","id":"2306.03236","version":1},"attestation_state":"computed","paper":{"title":"A Study of Global and Episodic Bonuses for Exploration in Contextual MDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Mikael Henaff, Minqi Jiang, Roberta Raileanu","submitted_at":"2023-06-05T20:45:30Z","abstract_excerpt":"Exploration in environments which differ across episodes has received increasing attention in recent years. Current methods use some combination of global novelty bonuses, computed using the agent's entire training experience, and \\textit{episodic novelty bonuses}, computed using only experience from the current episode. However, the use of these two types of bonuses has been ad-hoc and poorly understood. In this work, we shed light on the behavior of these two types of bonuses through controlled experiments on easily interpretable tasks as well as challenging pixel-based settings. We find tha"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.03236","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-06-05T20:45:30Z","cross_cats_sorted":[],"title_canon_sha256":"feb88fc713088b0c26e7d37b777a3d0826d7b2126ff6e3865eaf9f34aeebe93c","abstract_canon_sha256":"226554b447a77c407968250ec5a3d099012674b81f4de8914e55122fa5a51a6c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:17:55.427833Z","signature_b64":"TlDLOw1j4Rqe3i1DsGEX7B4BLU7WnVA21FsZB+wYZRg4kgMU9CQO6DRqZ6sP5csjjEIi4ySuds/nowkqxOxyBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"004902911ae38cc9c7909784231b854c8ef0d32a3a8b4908a49410d50633c668","last_reissued_at":"2026-07-05T06:17:55.427425Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:17:55.427425Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Study of Global and Episodic Bonuses for Exploration in Contextual MDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Mikael Henaff, Minqi Jiang, Roberta Raileanu","submitted_at":"2023-06-05T20:45:30Z","abstract_excerpt":"Exploration in environments which differ across episodes has received increasing attention in recent years. Current methods use some combination of global novelty bonuses, computed using the agent's entire training experience, and \\textit{episodic novelty bonuses}, computed using only experience from the current episode. However, the use of these two types of bonuses has been ad-hoc and poorly understood. In this work, we shed light on the behavior of these two types of bonuses through controlled experiments on easily interpretable tasks as well as challenging pixel-based settings. We find tha"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.03236","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.03236/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.03236","created_at":"2026-07-05T06:17:55.427482+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.03236v1","created_at":"2026-07-05T06:17:55.427482+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.03236","created_at":"2026-07-05T06:17:55.427482+00:00"},{"alias_kind":"pith_short_12","alias_value":"ABEQFEI24OGM","created_at":"2026-07-05T06:17:55.427482+00:00"},{"alias_kind":"pith_short_16","alias_value":"ABEQFEI24OGMTR4Q","created_at":"2026-07-05T06:17:55.427482+00:00"},{"alias_kind":"pith_short_8","alias_value":"ABEQFEI2","created_at":"2026-07-05T06:17:55.427482+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.12627","citing_title":"Deep Reinforcement Learning with Hybrid Intrinsic Reward Model","ref_index":20,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS","json":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS.json","graph_json":"https://pith.science/api/pith-number/ABEQFEI24OGMTR4QS6CCGG4FJS/graph.json","events_json":"https://pith.science/api/pith-number/ABEQFEI24OGMTR4QS6CCGG4FJS/events.json","paper":"https://pith.science/paper/ABEQFEI2"},"agent_actions":{"view_html":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS","download_json":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS.json","view_paper":"https://pith.science/paper/ABEQFEI2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.03236&json=true","fetch_graph":"https://pith.science/api/pith-number/ABEQFEI24OGMTR4QS6CCGG4FJS/graph.json","fetch_events":"https://pith.science/api/pith-number/ABEQFEI24OGMTR4QS6CCGG4FJS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS/action/storage_attestation","attest_author":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS/action/author_attestation","sign_citation":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS/action/citation_signature","submit_replication":"https://pith.science/pith/ABEQFEI24OGMTR4QS6CCGG4FJS/action/replication_record"}},"created_at":"2026-07-05T06:17:55.427482+00:00","updated_at":"2026-07-05T06:17:55.427482+00:00"}