{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:L4NNST7NJQESD5TUAGZXI6JZAS","short_pith_number":"pith:L4NNST7N","schema_version":"1.0","canonical_sha256":"5f1ad94fed4c0921f67401b374793904ba6fc5191fb28bff41b9b8c56dfbb53b","source":{"kind":"arxiv","id":"2211.10515","version":2},"attestation_state":"computed","paper":{"title":"Curiosity in Hindsight: Intrinsic Exploration in Stochastic Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Corentin Tallec, Daniel Jarrett, Florent Altch\\'e, Michal Valko, R\\'emi Munos, Thomas Mesnard","submitted_at":"2022-11-18T21:49:53Z","abstract_excerpt":"Consider the problem of exploration in sparse-reward or reward-free environments, such as in Montezuma's Revenge. In the curiosity-driven paradigm, the agent is rewarded for how much each realized outcome differs from their predicted outcome. But using predictive error as intrinsic motivation is fragile in stochastic environments, as the agent may become trapped by high-entropy areas of the state-action space, such as a \"noisy TV\". In this work, we study a natural solution derived from structural causal models of the world: Our key idea is to learn representations of the future that capture pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.10515","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2022-11-18T21:49:53Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"72baa0440eaddc7aa2e87080c0579b26a9931434a208f6022d2fd4375cccd1a0","abstract_canon_sha256":"84c22d5e3c16ba3949581b9bf83e351f9a10d6bd9ecb89a3920906fa7b2adf9e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:32:52.788936Z","signature_b64":"UDbv2xZL97GTknbHHiVijOS5ZjJEgJZRpBWbWQZzI566mHF8r/5w9MJGQLNx9jDLLcxZgj0jkoz7q69U+MOvBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5f1ad94fed4c0921f67401b374793904ba6fc5191fb28bff41b9b8c56dfbb53b","last_reissued_at":"2026-07-05T06:32:52.788544Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:32:52.788544Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Curiosity in Hindsight: Intrinsic Exploration in Stochastic Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Corentin Tallec, Daniel Jarrett, Florent Altch\\'e, Michal Valko, R\\'emi Munos, Thomas Mesnard","submitted_at":"2022-11-18T21:49:53Z","abstract_excerpt":"Consider the problem of exploration in sparse-reward or reward-free environments, such as in Montezuma's Revenge. In the curiosity-driven paradigm, the agent is rewarded for how much each realized outcome differs from their predicted outcome. But using predictive error as intrinsic motivation is fragile in stochastic environments, as the agent may become trapped by high-entropy areas of the state-action space, such as a \"noisy TV\". In this work, we study a natural solution derived from structural causal models of the world: Our key idea is to learn representations of the future that capture pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.10515","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.10515/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.10515","created_at":"2026-07-05T06:32:52.788601+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.10515v2","created_at":"2026-07-05T06:32:52.788601+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.10515","created_at":"2026-07-05T06:32:52.788601+00:00"},{"alias_kind":"pith_short_12","alias_value":"L4NNST7NJQES","created_at":"2026-07-05T06:32:52.788601+00:00"},{"alias_kind":"pith_short_16","alias_value":"L4NNST7NJQESD5TU","created_at":"2026-07-05T06:32:52.788601+00:00"},{"alias_kind":"pith_short_8","alias_value":"L4NNST7N","created_at":"2026-07-05T06:32:52.788601+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS","json":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS.json","graph_json":"https://pith.science/api/pith-number/L4NNST7NJQESD5TUAGZXI6JZAS/graph.json","events_json":"https://pith.science/api/pith-number/L4NNST7NJQESD5TUAGZXI6JZAS/events.json","paper":"https://pith.science/paper/L4NNST7N"},"agent_actions":{"view_html":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS","download_json":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS.json","view_paper":"https://pith.science/paper/L4NNST7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.10515&json=true","fetch_graph":"https://pith.science/api/pith-number/L4NNST7NJQESD5TUAGZXI6JZAS/graph.json","fetch_events":"https://pith.science/api/pith-number/L4NNST7NJQESD5TUAGZXI6JZAS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS/action/storage_attestation","attest_author":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS/action/author_attestation","sign_citation":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS/action/citation_signature","submit_replication":"https://pith.science/pith/L4NNST7NJQESD5TUAGZXI6JZAS/action/replication_record"}},"created_at":"2026-07-05T06:32:52.788601+00:00","updated_at":"2026-07-05T06:32:52.788601+00:00"}