{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KQWHKGEHBIBZWKO6ZDL2SYDKBC","short_pith_number":"pith:KQWHKGEH","schema_version":"1.0","canonical_sha256":"542c7518870a039b29dec8d7a9606a08b50a638fb184fbc51d67ffbc277c7488","source":{"kind":"arxiv","id":"2309.13915","version":2},"attestation_state":"computed","paper":{"title":"Sample Complexity of Neural Policy Mirror Descent for Policy Optimization on Low-Dimensional Manifolds","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Minshuo Chen, Tuo Zhao, Xiang Ji, Zhenghao Xu","submitted_at":"2023-09-25T07:31:22Z","abstract_excerpt":"Policy gradient methods equipped with deep neural networks have achieved great success in solving high-dimensional reinforcement learning (RL) problems. However, current analyses cannot explain why they are resistant to the curse of dimensionality. In this work, we study the sample complexity of the neural policy mirror descent (NPMD) algorithm with deep convolutional neural networks (CNN). Motivated by the empirical observation that many high-dimensional environments have state spaces possessing low-dimensional structures, such as those taking images as states, we consider the state space to "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.13915","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-25T07:31:22Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"304e4d23c0130579c9af449cf85931f9c431240584a6a3e7915a078319e10c40","abstract_canon_sha256":"f68481401a7a55b9fa0e25072ac5f3b75707fcf6ba4b8f018179cab06accc409"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:33:20.013176Z","signature_b64":"0J5eXrbsaQDroo+5RPvJbPoX/tklKQPU0V+ELnw1FcHMI+LvttCWPThpgoE5j/ZovukCWUSAWQ3ErN+vhkCVCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"542c7518870a039b29dec8d7a9606a08b50a638fb184fbc51d67ffbc277c7488","last_reissued_at":"2026-07-05T07:33:20.012703Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:33:20.012703Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sample Complexity of Neural Policy Mirror Descent for Policy Optimization on Low-Dimensional Manifolds","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Mengdi Wang, Minshuo Chen, Tuo Zhao, Xiang Ji, Zhenghao Xu","submitted_at":"2023-09-25T07:31:22Z","abstract_excerpt":"Policy gradient methods equipped with deep neural networks have achieved great success in solving high-dimensional reinforcement learning (RL) problems. However, current analyses cannot explain why they are resistant to the curse of dimensionality. In this work, we study the sample complexity of the neural policy mirror descent (NPMD) algorithm with deep convolutional neural networks (CNN). Motivated by the empirical observation that many high-dimensional environments have state spaces possessing low-dimensional structures, such as those taking images as states, we consider the state space to "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.13915","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.13915/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.13915","created_at":"2026-07-05T07:33:20.012761+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.13915v2","created_at":"2026-07-05T07:33:20.012761+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.13915","created_at":"2026-07-05T07:33:20.012761+00:00"},{"alias_kind":"pith_short_12","alias_value":"KQWHKGEHBIBZ","created_at":"2026-07-05T07:33:20.012761+00:00"},{"alias_kind":"pith_short_16","alias_value":"KQWHKGEHBIBZWKO6","created_at":"2026-07-05T07:33:20.012761+00:00"},{"alias_kind":"pith_short_8","alias_value":"KQWHKGEH","created_at":"2026-07-05T07:33:20.012761+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC","json":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC.json","graph_json":"https://pith.science/api/pith-number/KQWHKGEHBIBZWKO6ZDL2SYDKBC/graph.json","events_json":"https://pith.science/api/pith-number/KQWHKGEHBIBZWKO6ZDL2SYDKBC/events.json","paper":"https://pith.science/paper/KQWHKGEH"},"agent_actions":{"view_html":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC","download_json":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC.json","view_paper":"https://pith.science/paper/KQWHKGEH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.13915&json=true","fetch_graph":"https://pith.science/api/pith-number/KQWHKGEHBIBZWKO6ZDL2SYDKBC/graph.json","fetch_events":"https://pith.science/api/pith-number/KQWHKGEHBIBZWKO6ZDL2SYDKBC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC/action/storage_attestation","attest_author":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC/action/author_attestation","sign_citation":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC/action/citation_signature","submit_replication":"https://pith.science/pith/KQWHKGEHBIBZWKO6ZDL2SYDKBC/action/replication_record"}},"created_at":"2026-07-05T07:33:20.012761+00:00","updated_at":"2026-07-05T07:33:20.012761+00:00"}