{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3J4QZLLTKMPDYRD6M3YQKUVG5C","short_pith_number":"pith:3J4QZLLT","schema_version":"1.0","canonical_sha256":"da790cad73531e3c447e66f10552a6e89811610a67e9b3944e9665ff649afbac","source":{"kind":"arxiv","id":"2405.19548","version":2},"attestation_state":"computed","paper":{"title":"RLeXplore: Accelerating Research in Intrinsically-Motivated Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Li, Glen Berseth, Mingqi Yuan, Roger Creus Castanyer, Wenjun Zeng, Xin Jin","submitted_at":"2024-05-29T22:23:20Z","abstract_excerpt":"Extrinsic rewards can effectively guide reinforcement learning (RL) agents in specific tasks. However, extrinsic rewards frequently fall short in complex environments due to the significant human effort needed for their design and annotation. This limitation underscores the necessity for intrinsic rewards, which offer auxiliary and dense signals and can enable agents to learn in an unsupervised manner. Although various intrinsic reward formulations have been proposed, their implementation and optimization details are insufficiently explored and lack standardization, thereby hindering research "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19548","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-29T22:23:20Z","cross_cats_sorted":[],"title_canon_sha256":"cf58f5dc32c4e18d033f53b5d72bbd863e3d8318897be34b480465cfe0ed8ef0","abstract_canon_sha256":"abc80109ede51f7de20d07a5ffd89947c6031fe43e67e41ef210a495cf423124"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:45.340545Z","signature_b64":"6+bQWdXDaNAdV80IsO3OpRznJyJwjG1lHeBPL3DFqSIiJdXJk9t2ifwgNwMSuiU2F22/+denCjU0A7j/ACPrAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da790cad73531e3c447e66f10552a6e89811610a67e9b3944e9665ff649afbac","last_reissued_at":"2026-07-05T10:53:45.339973Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:45.339973Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLeXplore: Accelerating Research in Intrinsically-Motivated Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Li, Glen Berseth, Mingqi Yuan, Roger Creus Castanyer, Wenjun Zeng, Xin Jin","submitted_at":"2024-05-29T22:23:20Z","abstract_excerpt":"Extrinsic rewards can effectively guide reinforcement learning (RL) agents in specific tasks. However, extrinsic rewards frequently fall short in complex environments due to the significant human effort needed for their design and annotation. This limitation underscores the necessity for intrinsic rewards, which offer auxiliary and dense signals and can enable agents to learn in an unsupervised manner. Although various intrinsic reward formulations have been proposed, their implementation and optimization details are insufficiently explored and lack standardization, thereby hindering research "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19548","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19548/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19548","created_at":"2026-07-05T10:53:45.340040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19548v2","created_at":"2026-07-05T10:53:45.340040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19548","created_at":"2026-07-05T10:53:45.340040+00:00"},{"alias_kind":"pith_short_12","alias_value":"3J4QZLLTKMPD","created_at":"2026-07-05T10:53:45.340040+00:00"},{"alias_kind":"pith_short_16","alias_value":"3J4QZLLTKMPDYRD6","created_at":"2026-07-05T10:53:45.340040+00:00"},{"alias_kind":"pith_short_8","alias_value":"3J4QZLLT","created_at":"2026-07-05T10:53:45.340040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18963","citing_title":"Online Reward-Punishment Learning from Fixed-Channel Perceptual Event Streams without Environment Rewards","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C","json":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C.json","graph_json":"https://pith.science/api/pith-number/3J4QZLLTKMPDYRD6M3YQKUVG5C/graph.json","events_json":"https://pith.science/api/pith-number/3J4QZLLTKMPDYRD6M3YQKUVG5C/events.json","paper":"https://pith.science/paper/3J4QZLLT"},"agent_actions":{"view_html":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C","download_json":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C.json","view_paper":"https://pith.science/paper/3J4QZLLT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19548&json=true","fetch_graph":"https://pith.science/api/pith-number/3J4QZLLTKMPDYRD6M3YQKUVG5C/graph.json","fetch_events":"https://pith.science/api/pith-number/3J4QZLLTKMPDYRD6M3YQKUVG5C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C/action/storage_attestation","attest_author":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C/action/author_attestation","sign_citation":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C/action/citation_signature","submit_replication":"https://pith.science/pith/3J4QZLLTKMPDYRD6M3YQKUVG5C/action/replication_record"}},"created_at":"2026-07-05T10:53:45.340040+00:00","updated_at":"2026-07-05T10:53:45.340040+00:00"}