{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BNOKT5WDUYHH33ETNWG74SSC33","short_pith_number":"pith:BNOKT5WD","schema_version":"1.0","canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","source":{"kind":"arxiv","id":"2508.18420","version":1},"attestation_state":"computed","paper":{"title":"LLM-Driven Intrinsic Motivation for Sparse Reward Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andr\\'e Quadros, Cassio Silva, Ronnie Alves","submitted_at":"2025-08-25T19:10:58Z","abstract_excerpt":"This paper explores the combination of two intrinsic motivation strategies to improve the efficiency of reinforcement learning (RL) agents in environments with extreme sparse rewards, where traditional learning struggles due to infrequent positive feedback. We propose integrating Variational State as Intrinsic Reward (VSIMR), which uses Variational AutoEncoders (VAEs) to reward state novelty, with an intrinsic reward approach derived from Large Language Models (LLMs). The LLMs leverage their pre-trained knowledge to generate reward signals based on environment and goal descriptions, guiding th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.18420","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-25T19:10:58Z","cross_cats_sorted":[],"title_canon_sha256":"3622b41995629898082b58810309326d844ed4f9f6c2f0bfd811cead689a3a84","abstract_canon_sha256":"5fc48c03b46663140652f1a923e1a1784d51bb29e2f72e9d4519a13b25347dbe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:23.983753Z","signature_b64":"gjjD29qTMRGFynH1WMDxKxoJur45eqo0sxc4+CkfgY1FUQ0r0q/HGg1KKpHMB03rBJNt7i5L5qsARoKFUJS2CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","last_reissued_at":"2026-07-05T11:59:23.983292Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:23.983292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-Driven Intrinsic Motivation for Sparse Reward Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andr\\'e Quadros, Cassio Silva, Ronnie Alves","submitted_at":"2025-08-25T19:10:58Z","abstract_excerpt":"This paper explores the combination of two intrinsic motivation strategies to improve the efficiency of reinforcement learning (RL) agents in environments with extreme sparse rewards, where traditional learning struggles due to infrequent positive feedback. We propose integrating Variational State as Intrinsic Reward (VSIMR), which uses Variational AutoEncoders (VAEs) to reward state novelty, with an intrinsic reward approach derived from Large Language Models (LLMs). The LLMs leverage their pre-trained knowledge to generate reward signals based on environment and goal descriptions, guiding th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.18420","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.18420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.18420","created_at":"2026-07-05T11:59:23.983365+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.18420v1","created_at":"2026-07-05T11:59:23.983365+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.18420","created_at":"2026-07-05T11:59:23.983365+00:00"},{"alias_kind":"pith_short_12","alias_value":"BNOKT5WDUYHH","created_at":"2026-07-05T11:59:23.983365+00:00"},{"alias_kind":"pith_short_16","alias_value":"BNOKT5WDUYHH33ET","created_at":"2026-07-05T11:59:23.983365+00:00"},{"alias_kind":"pith_short_8","alias_value":"BNOKT5WD","created_at":"2026-07-05T11:59:23.983365+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33","json":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33.json","graph_json":"https://pith.science/api/pith-number/BNOKT5WDUYHH33ETNWG74SSC33/graph.json","events_json":"https://pith.science/api/pith-number/BNOKT5WDUYHH33ETNWG74SSC33/events.json","paper":"https://pith.science/paper/BNOKT5WD"},"agent_actions":{"view_html":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33","download_json":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33.json","view_paper":"https://pith.science/paper/BNOKT5WD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.18420&json=true","fetch_graph":"https://pith.science/api/pith-number/BNOKT5WDUYHH33ETNWG74SSC33/graph.json","fetch_events":"https://pith.science/api/pith-number/BNOKT5WDUYHH33ETNWG74SSC33/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/action/storage_attestation","attest_author":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/action/author_attestation","sign_citation":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/action/citation_signature","submit_replication":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/action/replication_record"}},"created_at":"2026-07-05T11:59:23.983365+00:00","updated_at":"2026-07-05T11:59:23.983365+00:00"}