{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:BNOKT5WDUYHH33ETNWG74SSC33","short_pith_number":"pith:BNOKT5WD","canonical_record":{"source":{"id":"2508.18420","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-25T19:10:58Z","cross_cats_sorted":[],"title_canon_sha256":"3622b41995629898082b58810309326d844ed4f9f6c2f0bfd811cead689a3a84","abstract_canon_sha256":"5fc48c03b46663140652f1a923e1a1784d51bb29e2f72e9d4519a13b25347dbe"},"schema_version":"1.0"},"canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","source":{"kind":"arxiv","id":"2508.18420","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.18420","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"arxiv_version","alias_value":"2508.18420v1","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.18420","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"pith_short_12","alias_value":"BNOKT5WDUYHH","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"pith_short_16","alias_value":"BNOKT5WDUYHH33ET","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"pith_short_8","alias_value":"BNOKT5WD","created_at":"2026-07-05T11:59:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:BNOKT5WDUYHH33ETNWG74SSC33","target":"record","payload":{"canonical_record":{"source":{"id":"2508.18420","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-25T19:10:58Z","cross_cats_sorted":[],"title_canon_sha256":"3622b41995629898082b58810309326d844ed4f9f6c2f0bfd811cead689a3a84","abstract_canon_sha256":"5fc48c03b46663140652f1a923e1a1784d51bb29e2f72e9d4519a13b25347dbe"},"schema_version":"1.0"},"canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:23.983753Z","signature_b64":"gjjD29qTMRGFynH1WMDxKxoJur45eqo0sxc4+CkfgY1FUQ0r0q/HGg1KKpHMB03rBJNt7i5L5qsARoKFUJS2CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","last_reissued_at":"2026-07-05T11:59:23.983292Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:23.983292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2508.18420","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:59:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0ASzD8uZIhj/PK6nUFNPaxsI2srPNbBczBoBcj4mif67WbyaVoJ2aX5XkTWMEuk/aywpmxKWKhLt3O0FGjwFBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T21:45:53.522351Z"},"content_sha256":"4f5e05ea3c3ec2fbbba4d884d6dfa9f2e7decef92e451a8075f1a3835e06b01d","schema_version":"1.0","event_id":"sha256:4f5e05ea3c3ec2fbbba4d884d6dfa9f2e7decef92e451a8075f1a3835e06b01d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:BNOKT5WDUYHH33ETNWG74SSC33","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"LLM-Driven Intrinsic Motivation for Sparse Reward Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andr\\'e Quadros, Cassio Silva, Ronnie Alves","submitted_at":"2025-08-25T19:10:58Z","abstract_excerpt":"This paper explores the combination of two intrinsic motivation strategies to improve the efficiency of reinforcement learning (RL) agents in environments with extreme sparse rewards, where traditional learning struggles due to infrequent positive feedback. We propose integrating Variational State as Intrinsic Reward (VSIMR), which uses Variational AutoEncoders (VAEs) to reward state novelty, with an intrinsic reward approach derived from Large Language Models (LLMs). The LLMs leverage their pre-trained knowledge to generate reward signals based on environment and goal descriptions, guiding th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.18420","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.18420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:59:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WVWeQvihn81wjI26DFG63jXRfOZX0ZXPiDg5BCT/27E4bK6mNmpAguoG6d0FsBPIS7OR+r/58NYCzifdFE0dBA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T21:45:53.522736Z"},"content_sha256":"4ee37c9c5e234b7e6c9571d3b52e61fd3d5312adc5879f57c5fcdfcf81bd6661","schema_version":"1.0","event_id":"sha256:4ee37c9c5e234b7e6c9571d3b52e61fd3d5312adc5879f57c5fcdfcf81bd6661"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/bundle.json","state_url":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/BNOKT5WDUYHH33ETNWG74SSC33/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T21:45:53Z","links":{"resolver":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33","bundle":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/bundle.json","state":"https://pith.science/pith/BNOKT5WDUYHH33ETNWG74SSC33/state.json","well_known_bundle":"https://pith.science/.well-known/pith/BNOKT5WDUYHH33ETNWG74SSC33/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:BNOKT5WDUYHH33ETNWG74SSC33","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5fc48c03b46663140652f1a923e1a1784d51bb29e2f72e9d4519a13b25347dbe","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-25T19:10:58Z","title_canon_sha256":"3622b41995629898082b58810309326d844ed4f9f6c2f0bfd811cead689a3a84"},"schema_version":"1.0","source":{"id":"2508.18420","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.18420","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"arxiv_version","alias_value":"2508.18420v1","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.18420","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"pith_short_12","alias_value":"BNOKT5WDUYHH","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"pith_short_16","alias_value":"BNOKT5WDUYHH33ET","created_at":"2026-07-05T11:59:23Z"},{"alias_kind":"pith_short_8","alias_value":"BNOKT5WD","created_at":"2026-07-05T11:59:23Z"}],"graph_snapshots":[{"event_id":"sha256:4ee37c9c5e234b7e6c9571d3b52e61fd3d5312adc5879f57c5fcdfcf81bd6661","target":"graph","created_at":"2026-07-05T11:59:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2508.18420/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper explores the combination of two intrinsic motivation strategies to improve the efficiency of reinforcement learning (RL) agents in environments with extreme sparse rewards, where traditional learning struggles due to infrequent positive feedback. We propose integrating Variational State as Intrinsic Reward (VSIMR), which uses Variational AutoEncoders (VAEs) to reward state novelty, with an intrinsic reward approach derived from Large Language Models (LLMs). The LLMs leverage their pre-trained knowledge to generate reward signals based on environment and goal descriptions, guiding th","authors_text":"Andr\\'e Quadros, Cassio Silva, Ronnie Alves","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-25T19:10:58Z","title":"LLM-Driven Intrinsic Motivation for Sparse Reward Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.18420","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4f5e05ea3c3ec2fbbba4d884d6dfa9f2e7decef92e451a8075f1a3835e06b01d","target":"record","created_at":"2026-07-05T11:59:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5fc48c03b46663140652f1a923e1a1784d51bb29e2f72e9d4519a13b25347dbe","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-25T19:10:58Z","title_canon_sha256":"3622b41995629898082b58810309326d844ed4f9f6c2f0bfd811cead689a3a84"},"schema_version":"1.0","source":{"id":"2508.18420","kind":"arxiv","version":1}},"canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"0b5ca9f6c3a60e7dec936d8dfe4a42deffecb0d18f1352b4fe67c1cddf1bc998","first_computed_at":"2026-07-05T11:59:23.983292Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:59:23.983292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"gjjD29qTMRGFynH1WMDxKxoJur45eqo0sxc4+CkfgY1FUQ0r0q/HGg1KKpHMB03rBJNt7i5L5qsARoKFUJS2CA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:59:23.983753Z","signed_message":"canonical_sha256_bytes"},"source_id":"2508.18420","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4f5e05ea3c3ec2fbbba4d884d6dfa9f2e7decef92e451a8075f1a3835e06b01d","sha256:4ee37c9c5e234b7e6c9571d3b52e61fd3d5312adc5879f57c5fcdfcf81bd6661"],"state_sha256":"c690ae24f789b73507d0b0306d7355308ec5ed406948a95a5fda9061e8389d1a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"lma7RCd+v5oyjZEZa2b8qI+PH6kQo5BCbG2me9VhLJpZTTNQ3kdRXcfnp2A3M3iSipliaVRuoKPdoTIQuzL5BQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T21:45:53.525692Z","bundle_sha256":"6de06cc7a7b3991820642c784528deb61d9595b16f06db12aa87290ff12b3906"}}