{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:VDYRYRNX5OW5TMQIIBRIGLCPCY","short_pith_number":"pith:VDYRYRNX","schema_version":"1.0","canonical_sha256":"a8f11c45b7ebadd9b2084062832c4f161f1d50c668b2add4751b926a80e5c167","source":{"kind":"arxiv","id":"1805.02785","version":3},"attestation_state":"computed","paper":{"title":"Fast Online Exact Solutions for Deterministic MDPs with Sparse Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Joshua R. Bertram, Peng Wei, Xuxi Yang","submitted_at":"2018-05-08T00:12:43Z","abstract_excerpt":"Markov Decision Processes (MDPs) are a mathematical framework for modeling sequential decision making under uncertainty. The classical approaches for solving MDPs are well known and have been widely studied, some of which rely on approximation techniques to solve MDPs with large state space and/or action space. However, most of these classical solution approaches and their approximation techniques still take much computation time to converge and usually must be re-computed if the reward function is changed. This paper introduces a novel alternative approach for exactly and efficiently solving "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1805.02785","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-05-08T00:12:43Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"3c01d2e0170b8d9c46e3278b5d23c74042be45c2a373d384d87e4a50fe936683","abstract_canon_sha256":"921b2233fbda151dfcc79413e04a5cd094235c7d4be7d8d30d4ca61fd721a60c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:15:46.323228Z","signature_b64":"YMh97du3ZDLBgD/I8u48+olEyCPrRDpnenr1PGP0jxU+S7LFsZOmI5kAE1IDbTV7S79nzwliXghii4NsQAwhBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a8f11c45b7ebadd9b2084062832c4f161f1d50c668b2add4751b926a80e5c167","last_reissued_at":"2026-05-18T00:15:46.322616Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:15:46.322616Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fast Online Exact Solutions for Deterministic MDPs with Sparse Rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Joshua R. Bertram, Peng Wei, Xuxi Yang","submitted_at":"2018-05-08T00:12:43Z","abstract_excerpt":"Markov Decision Processes (MDPs) are a mathematical framework for modeling sequential decision making under uncertainty. The classical approaches for solving MDPs are well known and have been widely studied, some of which rely on approximation techniques to solve MDPs with large state space and/or action space. However, most of these classical solution approaches and their approximation techniques still take much computation time to converge and usually must be re-computed if the reward function is changed. This paper introduces a novel alternative approach for exactly and efficiently solving "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1805.02785","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1805.02785","created_at":"2026-05-18T00:15:46.322721+00:00"},{"alias_kind":"arxiv_version","alias_value":"1805.02785v3","created_at":"2026-05-18T00:15:46.322721+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1805.02785","created_at":"2026-05-18T00:15:46.322721+00:00"},{"alias_kind":"pith_short_12","alias_value":"VDYRYRNX5OW5","created_at":"2026-05-18T12:32:59.047623+00:00"},{"alias_kind":"pith_short_16","alias_value":"VDYRYRNX5OW5TMQI","created_at":"2026-05-18T12:32:59.047623+00:00"},{"alias_kind":"pith_short_8","alias_value":"VDYRYRNX","created_at":"2026-05-18T12:32:59.047623+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.14093","citing_title":"Physics-Informed Reward Machines","ref_index":2020,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY","json":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY.json","graph_json":"https://pith.science/api/pith-number/VDYRYRNX5OW5TMQIIBRIGLCPCY/graph.json","events_json":"https://pith.science/api/pith-number/VDYRYRNX5OW5TMQIIBRIGLCPCY/events.json","paper":"https://pith.science/paper/VDYRYRNX"},"agent_actions":{"view_html":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY","download_json":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY.json","view_paper":"https://pith.science/paper/VDYRYRNX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1805.02785&json=true","fetch_graph":"https://pith.science/api/pith-number/VDYRYRNX5OW5TMQIIBRIGLCPCY/graph.json","fetch_events":"https://pith.science/api/pith-number/VDYRYRNX5OW5TMQIIBRIGLCPCY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY/action/storage_attestation","attest_author":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY/action/author_attestation","sign_citation":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY/action/citation_signature","submit_replication":"https://pith.science/pith/VDYRYRNX5OW5TMQIIBRIGLCPCY/action/replication_record"}},"created_at":"2026-05-18T00:15:46.322721+00:00","updated_at":"2026-05-18T00:15:46.322721+00:00"}