{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2HFPKYXVONFVQ6LKYQVE73TFOD","short_pith_number":"pith:2HFPKYXV","schema_version":"1.0","canonical_sha256":"d1caf562f5734b58796ac42a4fee6570d87cf1492efce3c5bfe54ca101c5a987","source":{"kind":"arxiv","id":"2307.13372","version":2},"attestation_state":"computed","paper":{"title":"Submodular Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Manish Prajapat, Melanie N. Zeilinger, Mojm\\'ir Mutn\\'y","submitted_at":"2023-07-25T09:46:02Z","abstract_excerpt":"In reinforcement learning (RL), rewards of states are typically considered additive, and following the Markov assumption, they are $\\textit{independent}$ of states visited previously. In many important applications, such as coverage control, experiment design and informative path planning, rewards naturally have diminishing returns, i.e., their value decreases in light of similar states visited previously. To tackle this, we propose $\\textit{submodular RL}$ (SubRL), a paradigm which seeks to optimize more general, non-additive (and history-dependent) rewards modelled via submodular set functio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.13372","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-25T09:46:02Z","cross_cats_sorted":[],"title_canon_sha256":"9b59ebd8cdc9c7161872399ddaf7c0934d6392d92f0ad7b8b48b35c64b0e626e","abstract_canon_sha256":"a40468a4970ad2fa966535670f4959d16e0a75546c647fb304a0dc0d47c5e983"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:32.579934Z","signature_b64":"4+zpSd20DPY6smtocLutGMNQrY7DW7EtrksGGdCM7lGqB1rpCP2t8peuNBbj4veGJdAFww600/is3WQLKr92AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1caf562f5734b58796ac42a4fee6570d87cf1492efce3c5bfe54ca101c5a987","last_reissued_at":"2026-07-05T08:22:32.579350Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:32.579350Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Submodular Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andreas Krause, Manish Prajapat, Melanie N. Zeilinger, Mojm\\'ir Mutn\\'y","submitted_at":"2023-07-25T09:46:02Z","abstract_excerpt":"In reinforcement learning (RL), rewards of states are typically considered additive, and following the Markov assumption, they are $\\textit{independent}$ of states visited previously. In many important applications, such as coverage control, experiment design and informative path planning, rewards naturally have diminishing returns, i.e., their value decreases in light of similar states visited previously. To tackle this, we propose $\\textit{submodular RL}$ (SubRL), a paradigm which seeks to optimize more general, non-additive (and history-dependent) rewards modelled via submodular set functio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.13372","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.13372/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.13372","created_at":"2026-07-05T08:22:32.579424+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.13372v2","created_at":"2026-07-05T08:22:32.579424+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.13372","created_at":"2026-07-05T08:22:32.579424+00:00"},{"alias_kind":"pith_short_12","alias_value":"2HFPKYXVONFV","created_at":"2026-07-05T08:22:32.579424+00:00"},{"alias_kind":"pith_short_16","alias_value":"2HFPKYXVONFVQ6LK","created_at":"2026-07-05T08:22:32.579424+00:00"},{"alias_kind":"pith_short_8","alias_value":"2HFPKYXV","created_at":"2026-07-05T08:22:32.579424+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.13834","citing_title":"Scalable Submodular Policy Optimization via Pruned Submodularity Graph","ref_index":23,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD","json":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD.json","graph_json":"https://pith.science/api/pith-number/2HFPKYXVONFVQ6LKYQVE73TFOD/graph.json","events_json":"https://pith.science/api/pith-number/2HFPKYXVONFVQ6LKYQVE73TFOD/events.json","paper":"https://pith.science/paper/2HFPKYXV"},"agent_actions":{"view_html":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD","download_json":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD.json","view_paper":"https://pith.science/paper/2HFPKYXV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.13372&json=true","fetch_graph":"https://pith.science/api/pith-number/2HFPKYXVONFVQ6LKYQVE73TFOD/graph.json","fetch_events":"https://pith.science/api/pith-number/2HFPKYXVONFVQ6LKYQVE73TFOD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD/action/storage_attestation","attest_author":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD/action/author_attestation","sign_citation":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD/action/citation_signature","submit_replication":"https://pith.science/pith/2HFPKYXVONFVQ6LKYQVE73TFOD/action/replication_record"}},"created_at":"2026-07-05T08:22:32.579424+00:00","updated_at":"2026-07-05T08:22:32.579424+00:00"}