{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WY2EQUPKBDUWTDA7YQTUAKS2LF","short_pith_number":"pith:WY2EQUPK","schema_version":"1.0","canonical_sha256":"b6344851ea08e9698c1fc427402a5a5964d8389253010600c252530241be758f","source":{"kind":"arxiv","id":"2201.12518","version":4},"attestation_state":"computed","paper":{"title":"Zeroth-Order Actor-Critic: An Evolutionary Framework for Sequential Decision Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Guojian Zhan, Jiangtao Li, Jianyu Chen, Shengbo Eben Li, Sifa Zheng, Tao Zhang, Yao Lyu, Yuheng Lei","submitted_at":"2022-01-29T07:09:03Z","abstract_excerpt":"Evolutionary algorithms (EAs) have shown promise in solving sequential decision problems (SDPs) by simplifying them to static optimization problems and searching for the optimal policy parameters in a zeroth-order way. While these methods are highly versatile, they often suffer from high sample complexity due to their ignorance of the underlying temporal structures. In contrast, reinforcement learning (RL) methods typically formulate SDPs as Markov Decision Process (MDP). Although more sample efficient than EAs, RL methods are restricted to differentiable policies and prone to getting stuck in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.12518","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-29T07:09:03Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"title_canon_sha256":"2546748199d1859c07e63a9c5e62f3dbab8550ce741bbace935829d24679ab78","abstract_canon_sha256":"0317eabe5fede19a41210589c09617237d5229cb5b4a78ed4bf86e4be63b9846"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:59:39.963369Z","signature_b64":"LNkQvGmQiYqlcVlGGyUv0j+xY1yp19PiIVOaRE8DiHISu8cQkQyYc/xN8TFowET2dfMw2dfUODmwkIpuq5pWBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6344851ea08e9698c1fc427402a5a5964d8389253010600c252530241be758f","last_reissued_at":"2026-07-05T09:59:39.962931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:59:39.962931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zeroth-Order Actor-Critic: An Evolutionary Framework for Sequential Decision Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY"],"primary_cat":"cs.LG","authors_text":"Guojian Zhan, Jiangtao Li, Jianyu Chen, Shengbo Eben Li, Sifa Zheng, Tao Zhang, Yao Lyu, Yuheng Lei","submitted_at":"2022-01-29T07:09:03Z","abstract_excerpt":"Evolutionary algorithms (EAs) have shown promise in solving sequential decision problems (SDPs) by simplifying them to static optimization problems and searching for the optimal policy parameters in a zeroth-order way. While these methods are highly versatile, they often suffer from high sample complexity due to their ignorance of the underlying temporal structures. In contrast, reinforcement learning (RL) methods typically formulate SDPs as Markov Decision Process (MDP). Although more sample efficient than EAs, RL methods are restricted to differentiable policies and prone to getting stuck in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.12518","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.12518/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.12518","created_at":"2026-07-05T09:59:39.962987+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.12518v4","created_at":"2026-07-05T09:59:39.962987+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.12518","created_at":"2026-07-05T09:59:39.962987+00:00"},{"alias_kind":"pith_short_12","alias_value":"WY2EQUPKBDUW","created_at":"2026-07-05T09:59:39.962987+00:00"},{"alias_kind":"pith_short_16","alias_value":"WY2EQUPKBDUWTDA7","created_at":"2026-07-05T09:59:39.962987+00:00"},{"alias_kind":"pith_short_8","alias_value":"WY2EQUPK","created_at":"2026-07-05T09:59:39.962987+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF","json":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF.json","graph_json":"https://pith.science/api/pith-number/WY2EQUPKBDUWTDA7YQTUAKS2LF/graph.json","events_json":"https://pith.science/api/pith-number/WY2EQUPKBDUWTDA7YQTUAKS2LF/events.json","paper":"https://pith.science/paper/WY2EQUPK"},"agent_actions":{"view_html":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF","download_json":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF.json","view_paper":"https://pith.science/paper/WY2EQUPK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.12518&json=true","fetch_graph":"https://pith.science/api/pith-number/WY2EQUPKBDUWTDA7YQTUAKS2LF/graph.json","fetch_events":"https://pith.science/api/pith-number/WY2EQUPKBDUWTDA7YQTUAKS2LF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF/action/storage_attestation","attest_author":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF/action/author_attestation","sign_citation":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF/action/citation_signature","submit_replication":"https://pith.science/pith/WY2EQUPKBDUWTDA7YQTUAKS2LF/action/replication_record"}},"created_at":"2026-07-05T09:59:39.962987+00:00","updated_at":"2026-07-05T09:59:39.962987+00:00"}