{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:E7SOZCYG4RFWPM7ITXL5MVATQV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"18efd4692fe8267fa3ad47d097790e0913b0cf776793b44a216531290a160fca","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-05T19:48:03Z","title_canon_sha256":"e8e084a5ee2cac166466439cf257d0e4093f84de869a3f39ccf1134c22fd0560"},"schema_version":"1.0","source":{"id":"2307.02620","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2307.02620","created_at":"2026-07-05T08:09:43Z"},{"alias_kind":"arxiv_version","alias_value":"2307.02620v3","created_at":"2026-07-05T08:09:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.02620","created_at":"2026-07-05T08:09:43Z"},{"alias_kind":"pith_short_12","alias_value":"E7SOZCYG4RFW","created_at":"2026-07-05T08:09:43Z"},{"alias_kind":"pith_short_16","alias_value":"E7SOZCYG4RFWPM7I","created_at":"2026-07-05T08:09:43Z"},{"alias_kind":"pith_short_8","alias_value":"E7SOZCYG","created_at":"2026-07-05T08:09:43Z"}],"graph_snapshots":[{"event_id":"sha256:ace210b372cb9a27dcb35a07ce773472f7f5683ddcbb02370e6d974a0fa59541","target":"graph","created_at":"2026-07-05T08:09:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2307.02620/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) has been shown to learn sophisticated control policies for complex tasks including games, robotics, heating and cooling systems and text generation. The action-perception cycle in RL, however, generally assumes that a measurement of the state of the environment is available at each time step without a cost. In applications such as materials design, deep-sea and planetary robot exploration and medicine, however, there can be a high cost associated with measuring, or even approximating, the state of the environment. In this paper, we survey the recently growing litera","authors_text":"Colin Bellinger, Isaac Tamblyn, Mark Crowley","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-05T19:48:03Z","title":"Dynamic Observation Policies in Observation Cost-Sensitive Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.02620","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:579db40e2f13a75a28016c8eafd264a3aca6dc25d4b6a535049fc7523313afa0","target":"record","created_at":"2026-07-05T08:09:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"18efd4692fe8267fa3ad47d097790e0913b0cf776793b44a216531290a160fca","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-05T19:48:03Z","title_canon_sha256":"e8e084a5ee2cac166466439cf257d0e4093f84de869a3f39ccf1134c22fd0560"},"schema_version":"1.0","source":{"id":"2307.02620","kind":"arxiv","version":3}},"canonical_sha256":"27e4ec8b06e44b67b3e89dd7d654138571acf55bb73ddd632ae35c434a2b9827","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"27e4ec8b06e44b67b3e89dd7d654138571acf55bb73ddd632ae35c434a2b9827","first_computed_at":"2026-07-05T08:09:43.301704Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:09:43.301704Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"rT9F1N0I7GO9nR6ZpfqCRILuwcRHwgJfRjnDC1bJ5EqL2qO8PvPYSrzj1UyXc4lS/iuUep6Qd5lKE+wyqnPWBA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:09:43.302334Z","signed_message":"canonical_sha256_bytes"},"source_id":"2307.02620","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:579db40e2f13a75a28016c8eafd264a3aca6dc25d4b6a535049fc7523313afa0","sha256:ace210b372cb9a27dcb35a07ce773472f7f5683ddcbb02370e6d974a0fa59541"],"state_sha256":"a54c2a465c10734f05227739703213528ebc0d8a6cab72e8898bc44881f4878b"}