{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:NXW72IBWHINQAR6ZESJI2AYYAL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"95d636be497b3f82582573333585c21475b3df62ae01d5add4d30f7ff97aff28","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-23T23:31:21Z","title_canon_sha256":"bf2ce0f01dd3f81f6b74a566fe638c9814df1ce22d440b201fa7e76272783ccc"},"schema_version":"1.0","source":{"id":"1910.10840","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1910.10840","created_at":"2026-07-05T04:29:06Z"},{"alias_kind":"arxiv_version","alias_value":"1910.10840v1","created_at":"2026-07-05T04:29:06Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.10840","created_at":"2026-07-05T04:29:06Z"},{"alias_kind":"pith_short_12","alias_value":"NXW72IBWHINQ","created_at":"2026-07-05T04:29:06Z"},{"alias_kind":"pith_short_16","alias_value":"NXW72IBWHINQAR6Z","created_at":"2026-07-05T04:29:06Z"},{"alias_kind":"pith_short_8","alias_value":"NXW72IBW","created_at":"2026-07-05T04:29:06Z"}],"graph_snapshots":[{"event_id":"sha256:03f720f1f93a4e200fd2c23c69779005404ebfe51874085a33b229eaee6333bc","target":"graph","created_at":"2026-07-05T04:29:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/1910.10840/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning enables to train an agent via interaction with the environment. However, in the majority of real-world scenarios, the extrinsic feedback is sparse or not sufficient, thus intrinsic reward formulations are needed to successfully train the agent. This work investigates and extends the paradigm of curiosity-driven exploration. First, a probabilistic approach is taken to exploit the advantages of the attention mechanism, which is successfully applied in other domains of Deep Learning. Combining them, we propose new methods, such as AttA2C, an extension of the Actor-Critic fr","authors_text":"M\\'arton Szemenyei, Patrik Reizinger","cross_cats":["cs.AI","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-23T23:31:21Z","title":"Attention-based Curiosity-driven Exploration in Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.10840","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4fc85985b50469c5d8e593c8d7dea1867ca8943a01d9b2537844a28d794dc7dd","target":"record","created_at":"2026-07-05T04:29:06Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"95d636be497b3f82582573333585c21475b3df62ae01d5add4d30f7ff97aff28","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-23T23:31:21Z","title_canon_sha256":"bf2ce0f01dd3f81f6b74a566fe638c9814df1ce22d440b201fa7e76272783ccc"},"schema_version":"1.0","source":{"id":"1910.10840","kind":"arxiv","version":1}},"canonical_sha256":"6dedfd20363a1b0047d924928d031802d4c7f85c427566e0d4fa957cbf8fa9d6","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6dedfd20363a1b0047d924928d031802d4c7f85c427566e0d4fa957cbf8fa9d6","first_computed_at":"2026-07-05T04:29:06.466575Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:29:06.466575Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ob81SSiH+i5bPcvlWAPslyK4YIE81HiiOBDWXSxbTK9Mb0+28mmoAOF6scS6hsPLaY9eR1J4k4kJH+29ba+CAg==","signature_status":"signed_v1","signed_at":"2026-07-05T04:29:06.466965Z","signed_message":"canonical_sha256_bytes"},"source_id":"1910.10840","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4fc85985b50469c5d8e593c8d7dea1867ca8943a01d9b2537844a28d794dc7dd","sha256:03f720f1f93a4e200fd2c23c69779005404ebfe51874085a33b229eaee6333bc"],"state_sha256":"a701ea10060411106862ee73f4917033af925852d91045b88442c08cd3aa1e3b"}