{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TSX5EV42RQLGFM5UE4HMHD7QI7","short_pith_number":"pith:TSX5EV42","schema_version":"1.0","canonical_sha256":"9cafd2579a8c1662b3b4270ec38ff047fa94be962621ac45aefcfb6e1f2b5e58","source":{"kind":"arxiv","id":"2301.04939","version":1},"attestation_state":"computed","paper":{"title":"Safe Policy Improvement for POMDPs via Finite-State Controllers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Marnix Suilen, Nils Jansen, Thiago D. Sim\\~ao","submitted_at":"2023-01-12T11:22:54Z","abstract_excerpt":"We study safe policy improvement (SPI) for partially observable Markov decision processes (POMDPs). SPI is an offline reinforcement learning (RL) problem that assumes access to (1) historical data about an environment, and (2) the so-called behavior policy that previously generated this data by interacting with the environment. SPI methods neither require access to a model nor the environment itself, and aim to reliably improve the behavior policy in an offline manner. Existing methods make the strong assumption that the environment is fully observable. In our novel approach to the SPI problem"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.04939","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-01-12T11:22:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"fadb6a69df40e3b756714cfe44c8ed428f60285ebb18e7aed08a52eafdee8270","abstract_canon_sha256":"63ff9d2ffdde54b639e59578766bb2ec2a75acb3ac5903dec275d2a17dd44b23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:32:37.911027Z","signature_b64":"wIxI+IajzQQEayx0Wn/9x6Pn002QP8sJUqRDXpGevOyRf7PQCdbXyz3e5OE8ipFCtFJdWuImfYpmSyod82LFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9cafd2579a8c1662b3b4270ec38ff047fa94be962621ac45aefcfb6e1f2b5e58","last_reissued_at":"2026-07-05T05:32:37.910626Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:32:37.910626Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Policy Improvement for POMDPs via Finite-State Controllers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Marnix Suilen, Nils Jansen, Thiago D. Sim\\~ao","submitted_at":"2023-01-12T11:22:54Z","abstract_excerpt":"We study safe policy improvement (SPI) for partially observable Markov decision processes (POMDPs). SPI is an offline reinforcement learning (RL) problem that assumes access to (1) historical data about an environment, and (2) the so-called behavior policy that previously generated this data by interacting with the environment. SPI methods neither require access to a model nor the environment itself, and aim to reliably improve the behavior policy in an offline manner. Existing methods make the strong assumption that the environment is fully observable. In our novel approach to the SPI problem"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.04939","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.04939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.04939","created_at":"2026-07-05T05:32:37.910680+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.04939v1","created_at":"2026-07-05T05:32:37.910680+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.04939","created_at":"2026-07-05T05:32:37.910680+00:00"},{"alias_kind":"pith_short_12","alias_value":"TSX5EV42RQLG","created_at":"2026-07-05T05:32:37.910680+00:00"},{"alias_kind":"pith_short_16","alias_value":"TSX5EV42RQLGFM5U","created_at":"2026-07-05T05:32:37.910680+00:00"},{"alias_kind":"pith_short_8","alias_value":"TSX5EV42","created_at":"2026-07-05T05:32:37.910680+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10870","citing_title":"Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7","json":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7.json","graph_json":"https://pith.science/api/pith-number/TSX5EV42RQLGFM5UE4HMHD7QI7/graph.json","events_json":"https://pith.science/api/pith-number/TSX5EV42RQLGFM5UE4HMHD7QI7/events.json","paper":"https://pith.science/paper/TSX5EV42"},"agent_actions":{"view_html":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7","download_json":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7.json","view_paper":"https://pith.science/paper/TSX5EV42","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.04939&json=true","fetch_graph":"https://pith.science/api/pith-number/TSX5EV42RQLGFM5UE4HMHD7QI7/graph.json","fetch_events":"https://pith.science/api/pith-number/TSX5EV42RQLGFM5UE4HMHD7QI7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7/action/storage_attestation","attest_author":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7/action/author_attestation","sign_citation":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7/action/citation_signature","submit_replication":"https://pith.science/pith/TSX5EV42RQLGFM5UE4HMHD7QI7/action/replication_record"}},"created_at":"2026-07-05T05:32:37.910680+00:00","updated_at":"2026-07-05T05:32:37.910680+00:00"}