{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5T3BLXIMGCWP5KTHHYG5W4IBJI","short_pith_number":"pith:5T3BLXIM","schema_version":"1.0","canonical_sha256":"ecf615dd0c30acfeaa673e0ddb71014a27d17f25d18c5a3c95f4ca694a5cb4ad","source":{"kind":"arxiv","id":"2106.10566","version":2},"attestation_state":"computed","paper":{"title":"Scalable Safety-Critical Policy Evaluation with Accelerated Rare Event Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ding Zhao, Fengpei Li, Henry Lam, Jiacheng Zhu, Kentaro Oguchi, Mengdi Xu, Peide Huang, Xuewei Qi, Zhiyuan Huang","submitted_at":"2021-06-19T20:03:26Z","abstract_excerpt":"Evaluating rare but high-stakes events is one of the main challenges in obtaining reliable reinforcement learning policies, especially in large or infinite state/action spaces where limited scalability dictates a prohibitively large number of testing iterations. On the other hand, a biased or inaccurate policy evaluation in a safety-critical system could potentially cause unexpected catastrophic failures during deployment. This paper proposes the Accelerated Policy Evaluation (APE) method, which simultaneously uncovers rare events and estimates the rare event probability in Markov decision pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.10566","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-19T20:03:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"22e48ef31f6c5b55b67edd39036b18a75bfffe24a20a9a8928d1db5d1cdb4465","abstract_canon_sha256":"4c0b3230654b4c030866f4db332f8f9bd4c743628f83011978aa2d2457b7ebe1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:02:39.179817Z","signature_b64":"euD+HKbeviv4OqNZWyYoFjhxyrT3PTTY5opU81oP2itkGbSOLtsXUKniWXAmQnihefOwaVBjd3z5C+PII7buAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecf615dd0c30acfeaa673e0ddb71014a27d17f25d18c5a3c95f4ca694a5cb4ad","last_reissued_at":"2026-07-05T05:02:39.179328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:02:39.179328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scalable Safety-Critical Policy Evaluation with Accelerated Rare Event Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Ding Zhao, Fengpei Li, Henry Lam, Jiacheng Zhu, Kentaro Oguchi, Mengdi Xu, Peide Huang, Xuewei Qi, Zhiyuan Huang","submitted_at":"2021-06-19T20:03:26Z","abstract_excerpt":"Evaluating rare but high-stakes events is one of the main challenges in obtaining reliable reinforcement learning policies, especially in large or infinite state/action spaces where limited scalability dictates a prohibitively large number of testing iterations. On the other hand, a biased or inaccurate policy evaluation in a safety-critical system could potentially cause unexpected catastrophic failures during deployment. This paper proposes the Accelerated Policy Evaluation (APE) method, which simultaneously uncovers rare events and estimates the rare event probability in Markov decision pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.10566","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.10566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.10566","created_at":"2026-07-05T05:02:39.179393+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.10566v2","created_at":"2026-07-05T05:02:39.179393+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.10566","created_at":"2026-07-05T05:02:39.179393+00:00"},{"alias_kind":"pith_short_12","alias_value":"5T3BLXIMGCWP","created_at":"2026-07-05T05:02:39.179393+00:00"},{"alias_kind":"pith_short_16","alias_value":"5T3BLXIMGCWP5KTH","created_at":"2026-07-05T05:02:39.179393+00:00"},{"alias_kind":"pith_short_8","alias_value":"5T3BLXIM","created_at":"2026-07-05T05:02:39.179393+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI","json":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI.json","graph_json":"https://pith.science/api/pith-number/5T3BLXIMGCWP5KTHHYG5W4IBJI/graph.json","events_json":"https://pith.science/api/pith-number/5T3BLXIMGCWP5KTHHYG5W4IBJI/events.json","paper":"https://pith.science/paper/5T3BLXIM"},"agent_actions":{"view_html":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI","download_json":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI.json","view_paper":"https://pith.science/paper/5T3BLXIM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.10566&json=true","fetch_graph":"https://pith.science/api/pith-number/5T3BLXIMGCWP5KTHHYG5W4IBJI/graph.json","fetch_events":"https://pith.science/api/pith-number/5T3BLXIMGCWP5KTHHYG5W4IBJI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI/action/storage_attestation","attest_author":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI/action/author_attestation","sign_citation":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI/action/citation_signature","submit_replication":"https://pith.science/pith/5T3BLXIMGCWP5KTHHYG5W4IBJI/action/replication_record"}},"created_at":"2026-07-05T05:02:39.179393+00:00","updated_at":"2026-07-05T05:02:39.179393+00:00"}