{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:OW5YZ323YTEHVHDCPWUEGEAWBN","short_pith_number":"pith:OW5YZ323","canonical_record":{"source":{"id":"2407.12185","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-16T21:28:03Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"b32cc9bcf705419e2649fe748355fb83f0dd25e5705b0dbd387b47435459d8dd","abstract_canon_sha256":"c67cd1852c63d66bef2abaf2c3632a94113920a2a2e55d9abe1cd2cddc6c8ed2"},"schema_version":"1.0"},"canonical_sha256":"75bb8cef5bc4c87a9c627da84310160b41385b2582cefccbb9d9a2375bacf509","source":{"kind":"arxiv","id":"2407.12185","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.12185","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"arxiv_version","alias_value":"2407.12185v1","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12185","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"pith_short_12","alias_value":"OW5YZ323YTEH","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"pith_short_16","alias_value":"OW5YZ323YTEHVHDC","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"pith_short_8","alias_value":"OW5YZ323","created_at":"2026-07-05T08:46:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:OW5YZ323YTEHVHDCPWUEGEAWBN","target":"record","payload":{"canonical_record":{"source":{"id":"2407.12185","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-16T21:28:03Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"b32cc9bcf705419e2649fe748355fb83f0dd25e5705b0dbd387b47435459d8dd","abstract_canon_sha256":"c67cd1852c63d66bef2abaf2c3632a94113920a2a2e55d9abe1cd2cddc6c8ed2"},"schema_version":"1.0"},"canonical_sha256":"75bb8cef5bc4c87a9c627da84310160b41385b2582cefccbb9d9a2375bacf509","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:27.910676Z","signature_b64":"cXoJ3QuM8vR6e+ud1q51Z0GrVCvvEwf0QPn6jnwsb5njY+V5bkdzbNv40+DfCSj8pE03d8YiEbEqNGMnCSLpAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"75bb8cef5bc4c87a9c627da84310160b41385b2582cefccbb9d9a2375bacf509","last_reissued_at":"2026-07-05T08:46:27.910260Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:27.910260Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2407.12185","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:46:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"RWJHkABe56LdN78KLBhTujo6NDa787Yj9WDeEVxiY0JYsz3d4ocRzxLMxr9M3tdnxwpU6hp6AnO+PvvJPUp7DQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T21:39:49.975473Z"},"content_sha256":"56d0948f1f4005e6cb40a98453bbbd72dd51293af7669d12b2d32250e14b4695","schema_version":"1.0","event_id":"sha256:56d0948f1f4005e6cb40a98453bbbd72dd51293af7669d12b2d32250e14b4695"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:OW5YZ323YTEHVHDCPWUEGEAWBN","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Satisficing Exploration for Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Benjamin Van Roy, Dilip Arumugam, Ramki Gummadi, Saurabh Kumar","submitted_at":"2024-07-16T21:28:03Z","abstract_excerpt":"A default assumption in the design of reinforcement-learning algorithms is that a decision-making agent always explores to learn optimal behavior. In sufficiently complex environments that approach the vastness and scale of the real world, however, attaining optimal performance may in fact be an entirely intractable endeavor and an agent may seldom find itself in a position to complete the requisite exploration for identifying an optimal policy. Recent work has leveraged tools from information theory to design agents that deliberately forgo optimal solutions in favor of sufficiently-satisfying"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12185","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12185/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:46:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0I2WFX1+TOK0giHSusVpX/GFTrzEy5OlVRKeS223/3qogsSEf/Z+4QweqP4na2IQAIZNgizsTnrUPJvxC8qwBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T21:39:49.975962Z"},"content_sha256":"064bd5092292085005c36692ccfbf88ef8f2290b90110051c2468a8d6cc84c30","schema_version":"1.0","event_id":"sha256:064bd5092292085005c36692ccfbf88ef8f2290b90110051c2468a8d6cc84c30"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/OW5YZ323YTEHVHDCPWUEGEAWBN/bundle.json","state_url":"https://pith.science/pith/OW5YZ323YTEHVHDCPWUEGEAWBN/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/OW5YZ323YTEHVHDCPWUEGEAWBN/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T21:39:49Z","links":{"resolver":"https://pith.science/pith/OW5YZ323YTEHVHDCPWUEGEAWBN","bundle":"https://pith.science/pith/OW5YZ323YTEHVHDCPWUEGEAWBN/bundle.json","state":"https://pith.science/pith/OW5YZ323YTEHVHDCPWUEGEAWBN/state.json","well_known_bundle":"https://pith.science/.well-known/pith/OW5YZ323YTEHVHDCPWUEGEAWBN/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:OW5YZ323YTEHVHDCPWUEGEAWBN","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c67cd1852c63d66bef2abaf2c3632a94113920a2a2e55d9abe1cd2cddc6c8ed2","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-16T21:28:03Z","title_canon_sha256":"b32cc9bcf705419e2649fe748355fb83f0dd25e5705b0dbd387b47435459d8dd"},"schema_version":"1.0","source":{"id":"2407.12185","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.12185","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"arxiv_version","alias_value":"2407.12185v1","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12185","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"pith_short_12","alias_value":"OW5YZ323YTEH","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"pith_short_16","alias_value":"OW5YZ323YTEHVHDC","created_at":"2026-07-05T08:46:27Z"},{"alias_kind":"pith_short_8","alias_value":"OW5YZ323","created_at":"2026-07-05T08:46:27Z"}],"graph_snapshots":[{"event_id":"sha256:064bd5092292085005c36692ccfbf88ef8f2290b90110051c2468a8d6cc84c30","target":"graph","created_at":"2026-07-05T08:46:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2407.12185/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"A default assumption in the design of reinforcement-learning algorithms is that a decision-making agent always explores to learn optimal behavior. In sufficiently complex environments that approach the vastness and scale of the real world, however, attaining optimal performance may in fact be an entirely intractable endeavor and an agent may seldom find itself in a position to complete the requisite exploration for identifying an optimal policy. Recent work has leveraged tools from information theory to design agents that deliberately forgo optimal solutions in favor of sufficiently-satisfying","authors_text":"Benjamin Van Roy, Dilip Arumugam, Ramki Gummadi, Saurabh Kumar","cross_cats":["cs.AI","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-16T21:28:03Z","title":"Satisficing Exploration for Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12185","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:56d0948f1f4005e6cb40a98453bbbd72dd51293af7669d12b2d32250e14b4695","target":"record","created_at":"2026-07-05T08:46:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c67cd1852c63d66bef2abaf2c3632a94113920a2a2e55d9abe1cd2cddc6c8ed2","cross_cats_sorted":["cs.AI","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-16T21:28:03Z","title_canon_sha256":"b32cc9bcf705419e2649fe748355fb83f0dd25e5705b0dbd387b47435459d8dd"},"schema_version":"1.0","source":{"id":"2407.12185","kind":"arxiv","version":1}},"canonical_sha256":"75bb8cef5bc4c87a9c627da84310160b41385b2582cefccbb9d9a2375bacf509","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"75bb8cef5bc4c87a9c627da84310160b41385b2582cefccbb9d9a2375bacf509","first_computed_at":"2026-07-05T08:46:27.910260Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:46:27.910260Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"cXoJ3QuM8vR6e+ud1q51Z0GrVCvvEwf0QPn6jnwsb5njY+V5bkdzbNv40+DfCSj8pE03d8YiEbEqNGMnCSLpAw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:46:27.910676Z","signed_message":"canonical_sha256_bytes"},"source_id":"2407.12185","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:56d0948f1f4005e6cb40a98453bbbd72dd51293af7669d12b2d32250e14b4695","sha256:064bd5092292085005c36692ccfbf88ef8f2290b90110051c2468a8d6cc84c30"],"state_sha256":"7817cacfe9dab794f5cdb592fc835c84e028d843a8c1c45435aef7abb7ec70b4"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HGmll6XPgyuzAsjR1gpOJ2n9I0LOoNLFhFD9+dBPsb6cDzdsf6ETYaGLqEUi4UlqAvo3xLBaM19ya1F68wSlBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T21:39:49.979440Z","bundle_sha256":"dea1ece9d31f988ab51a54a0adc95ddd21394d7c2bf5afe4d8ca8c2992ea89a6"}}