{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:O4VOL625YSCLBB6V53K7YEOL6Y","short_pith_number":"pith:O4VOL625","schema_version":"1.0","canonical_sha256":"772ae5fb5dc484b087d5eed5fc11cbf6153681d49615e46361df61959b52200e","source":{"kind":"arxiv","id":"2003.01008","version":1},"attestation_state":"computed","paper":{"title":"Learning and Solving Regular Decision Processes","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Eden Abadi, Ronen I. Brafman","submitted_at":"2020-03-02T16:36:16Z","abstract_excerpt":"Regular Decision Processes (RDPs) are a recently introduced model that extends MDPs with non-Markovian dynamics and rewards. The non-Markovian behavior is restricted to depend on regular properties of the history. These can be specified using regular expressions or formulas in linear dynamic logic over finite traces. Fully specified RDPs can be solved by compiling them into an appropriate MDP. Learning RDPs from data is a challenging problem that has yet to be addressed, on which we focus in this paper. Our approach rests on a new representation for RDPs using Mealy Machines that emit a distri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.01008","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2020-03-02T16:36:16Z","cross_cats_sorted":[],"title_canon_sha256":"464ce242bf453d62f20ca248f102c55db82461d7c3677a2daa456fde539a1f6b","abstract_canon_sha256":"3ec93f534a4ba4e098f1bed6bd6232252a64a7dccebfe50a5442d64bf61b7d95"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:44:59.991373Z","signature_b64":"YO3xIyYRWGz0MtSTfYlNRqUu4rJO07IsqOrpSifDfd/7too6cvG01mOgJ2nyP9NkbdP25gyyHaObO9hk8hzTAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"772ae5fb5dc484b087d5eed5fc11cbf6153681d49615e46361df61959b52200e","last_reissued_at":"2026-07-05T00:44:59.990906Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:44:59.990906Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning and Solving Regular Decision Processes","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Eden Abadi, Ronen I. Brafman","submitted_at":"2020-03-02T16:36:16Z","abstract_excerpt":"Regular Decision Processes (RDPs) are a recently introduced model that extends MDPs with non-Markovian dynamics and rewards. The non-Markovian behavior is restricted to depend on regular properties of the history. These can be specified using regular expressions or formulas in linear dynamic logic over finite traces. Fully specified RDPs can be solved by compiling them into an appropriate MDP. Learning RDPs from data is a challenging problem that has yet to be addressed, on which we focus in this paper. Our approach rests on a new representation for RDPs using Mealy Machines that emit a distri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.01008","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.01008/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.01008","created_at":"2026-07-05T00:44:59.990964+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.01008v1","created_at":"2026-07-05T00:44:59.990964+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.01008","created_at":"2026-07-05T00:44:59.990964+00:00"},{"alias_kind":"pith_short_12","alias_value":"O4VOL625YSCL","created_at":"2026-07-05T00:44:59.990964+00:00"},{"alias_kind":"pith_short_16","alias_value":"O4VOL625YSCLBB6V","created_at":"2026-07-05T00:44:59.990964+00:00"},{"alias_kind":"pith_short_8","alias_value":"O4VOL625","created_at":"2026-07-05T00:44:59.990964+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00034","citing_title":"Bayesian updates from coalgebraic determinisation","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y","json":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y.json","graph_json":"https://pith.science/api/pith-number/O4VOL625YSCLBB6V53K7YEOL6Y/graph.json","events_json":"https://pith.science/api/pith-number/O4VOL625YSCLBB6V53K7YEOL6Y/events.json","paper":"https://pith.science/paper/O4VOL625"},"agent_actions":{"view_html":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y","download_json":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y.json","view_paper":"https://pith.science/paper/O4VOL625","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.01008&json=true","fetch_graph":"https://pith.science/api/pith-number/O4VOL625YSCLBB6V53K7YEOL6Y/graph.json","fetch_events":"https://pith.science/api/pith-number/O4VOL625YSCLBB6V53K7YEOL6Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y/action/storage_attestation","attest_author":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y/action/author_attestation","sign_citation":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y/action/citation_signature","submit_replication":"https://pith.science/pith/O4VOL625YSCLBB6V53K7YEOL6Y/action/replication_record"}},"created_at":"2026-07-05T00:44:59.990964+00:00","updated_at":"2026-07-05T00:44:59.990964+00:00"}