{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:HA5BQLJWPBAAIHUMND42AAHAWS","short_pith_number":"pith:HA5BQLJW","schema_version":"1.0","canonical_sha256":"383a182d367840041e8c68f9a000e0b4811186af75b94d376a40058bcdc6d258","source":{"kind":"arxiv","id":"2202.00817","version":2},"attestation_state":"computed","paper":{"title":"Do Differentiable Simulators Give Better Policy Gradients?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"H.J. Terry Suh, Kaiqing Zhang, Max Simchowitz, Russ Tedrake","submitted_at":"2022-02-02T00:12:28Z","abstract_excerpt":"Differentiable simulators promise faster computation time for reinforcement learning by replacing zeroth-order gradient estimates of a stochastic objective with an estimate based on first-order gradients. However, it is yet unclear what factors decide the performance of the two estimators on complex landscapes that involve long-horizon planning and control on physical systems, despite the crucial relevance of this question for the utility of differentiable simulators. We show that characteristics of certain physical systems, such as stiffness or discontinuities, may compromise the efficacy of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.00817","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-02T00:12:28Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"10b355ad9ab69339b4e3af1f2c0501bd1257646d733dd9172d9e975ae631ae73","abstract_canon_sha256":"4a3e8190505d80393350a42e5e36d89a1216d3fe55e24d285f979accb87410d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:50:14.203766Z","signature_b64":"FNBN6DC+Iolw1vXq+UYlNvLBLcGNzEjovFoKYVpCu8lDMUmNYBbXecT5IwGDnGRsJuldqJdd5Nn/yaemuLM4BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"383a182d367840041e8c68f9a000e0b4811186af75b94d376a40058bcdc6d258","last_reissued_at":"2026-07-05T04:50:14.203332Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:50:14.203332Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Differentiable Simulators Give Better Policy Gradients?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"H.J. Terry Suh, Kaiqing Zhang, Max Simchowitz, Russ Tedrake","submitted_at":"2022-02-02T00:12:28Z","abstract_excerpt":"Differentiable simulators promise faster computation time for reinforcement learning by replacing zeroth-order gradient estimates of a stochastic objective with an estimate based on first-order gradients. However, it is yet unclear what factors decide the performance of the two estimators on complex landscapes that involve long-horizon planning and control on physical systems, despite the crucial relevance of this question for the utility of differentiable simulators. We show that characteristics of certain physical systems, such as stiffness or discontinuities, may compromise the efficacy of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.00817","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.00817/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.00817","created_at":"2026-07-05T04:50:14.203390+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.00817v2","created_at":"2026-07-05T04:50:14.203390+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.00817","created_at":"2026-07-05T04:50:14.203390+00:00"},{"alias_kind":"pith_short_12","alias_value":"HA5BQLJWPBAA","created_at":"2026-07-05T04:50:14.203390+00:00"},{"alias_kind":"pith_short_16","alias_value":"HA5BQLJWPBAAIHUM","created_at":"2026-07-05T04:50:14.203390+00:00"},{"alias_kind":"pith_short_8","alias_value":"HA5BQLJW","created_at":"2026-07-05T04:50:14.203390+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26478","citing_title":"Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":200,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS","json":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS.json","graph_json":"https://pith.science/api/pith-number/HA5BQLJWPBAAIHUMND42AAHAWS/graph.json","events_json":"https://pith.science/api/pith-number/HA5BQLJWPBAAIHUMND42AAHAWS/events.json","paper":"https://pith.science/paper/HA5BQLJW"},"agent_actions":{"view_html":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS","download_json":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS.json","view_paper":"https://pith.science/paper/HA5BQLJW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.00817&json=true","fetch_graph":"https://pith.science/api/pith-number/HA5BQLJWPBAAIHUMND42AAHAWS/graph.json","fetch_events":"https://pith.science/api/pith-number/HA5BQLJWPBAAIHUMND42AAHAWS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS/action/storage_attestation","attest_author":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS/action/author_attestation","sign_citation":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS/action/citation_signature","submit_replication":"https://pith.science/pith/HA5BQLJWPBAAIHUMND42AAHAWS/action/replication_record"}},"created_at":"2026-07-05T04:50:14.203390+00:00","updated_at":"2026-07-05T04:50:14.203390+00:00"}