{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:4CDN42IQMHDBE5N7I7QFQBDG4I","short_pith_number":"pith:4CDN42IQ","schema_version":"1.0","canonical_sha256":"e086de691061c61275bf47e0580466e228abd966925be73383ba841bd1066923","source":{"kind":"arxiv","id":"2204.07137","version":1},"attestation_state":"computed","paper":{"title":"Accelerated Policy Learning with Parallel Differentiable Simulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.RO"],"primary_cat":"cs.LG","authors_text":"Animesh Garg, Fabio Ramos, Jie Xu, Miles Macklin, Viktor Makoviychuk, Wojciech Matusik, Yashraj Narang","submitted_at":"2022-04-14T17:46:26Z","abstract_excerpt":"Deep reinforcement learning can generate complex control policies, but requires large amounts of training data to work effectively. Recent work has attempted to address this issue by leveraging differentiable simulators. However, inherent problems such as local minima and exploding/vanishing numerical gradients prevent these methods from being generally applied to control tasks with complex contact-rich dynamics, such as humanoid locomotion in classical RL benchmarks. In this work we present a high-performance differentiable simulator and a new policy learning algorithm (SHAC) that can effecti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.07137","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-04-14T17:46:26Z","cross_cats_sorted":["cs.AI","cs.GR","cs.RO"],"title_canon_sha256":"4cbf1c389c5c42c49cc1ee219873cf9138ae27e355a85f1a270def374a41cebb","abstract_canon_sha256":"5c5f0606fced8323ecbeef9f3e7e22d9ca75765a00fcce0999ce95928e16c477"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:14:49.935998Z","signature_b64":"m5L9ibme+yvdnkzdxmN3npPQMsN5mRtMMChY1ygPp+eU0HQh/xms4UYGibMjqqKVent5n+j6sgtTknuvzvgcDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e086de691061c61275bf47e0580466e228abd966925be73383ba841bd1066923","last_reissued_at":"2026-07-05T04:14:49.935497Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:14:49.935497Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Accelerated Policy Learning with Parallel Differentiable Simulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.RO"],"primary_cat":"cs.LG","authors_text":"Animesh Garg, Fabio Ramos, Jie Xu, Miles Macklin, Viktor Makoviychuk, Wojciech Matusik, Yashraj Narang","submitted_at":"2022-04-14T17:46:26Z","abstract_excerpt":"Deep reinforcement learning can generate complex control policies, but requires large amounts of training data to work effectively. Recent work has attempted to address this issue by leveraging differentiable simulators. However, inherent problems such as local minima and exploding/vanishing numerical gradients prevent these methods from being generally applied to control tasks with complex contact-rich dynamics, such as humanoid locomotion in classical RL benchmarks. In this work we present a high-performance differentiable simulator and a new policy learning algorithm (SHAC) that can effecti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.07137","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.07137/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.07137","created_at":"2026-07-05T04:14:49.935555+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.07137v1","created_at":"2026-07-05T04:14:49.935555+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.07137","created_at":"2026-07-05T04:14:49.935555+00:00"},{"alias_kind":"pith_short_12","alias_value":"4CDN42IQMHDB","created_at":"2026-07-05T04:14:49.935555+00:00"},{"alias_kind":"pith_short_16","alias_value":"4CDN42IQMHDBE5N7","created_at":"2026-07-05T04:14:49.935555+00:00"},{"alias_kind":"pith_short_8","alias_value":"4CDN42IQ","created_at":"2026-07-05T04:14:49.935555+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.26478","citing_title":"Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2506.14009","citing_title":"GRaD-Nav++: Vision-Language Model Enabled Visual Drone Navigation with Gaussian Radiance Fields and Differentiable Dynamics","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10548","citing_title":"Simple but Stable, Fast and Safe: Achieve End-to-end Control by High-Fidelity Differentiable Simulation","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I","json":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I.json","graph_json":"https://pith.science/api/pith-number/4CDN42IQMHDBE5N7I7QFQBDG4I/graph.json","events_json":"https://pith.science/api/pith-number/4CDN42IQMHDBE5N7I7QFQBDG4I/events.json","paper":"https://pith.science/paper/4CDN42IQ"},"agent_actions":{"view_html":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I","download_json":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I.json","view_paper":"https://pith.science/paper/4CDN42IQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.07137&json=true","fetch_graph":"https://pith.science/api/pith-number/4CDN42IQMHDBE5N7I7QFQBDG4I/graph.json","fetch_events":"https://pith.science/api/pith-number/4CDN42IQMHDBE5N7I7QFQBDG4I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I/action/storage_attestation","attest_author":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I/action/author_attestation","sign_citation":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I/action/citation_signature","submit_replication":"https://pith.science/pith/4CDN42IQMHDBE5N7I7QFQBDG4I/action/replication_record"}},"created_at":"2026-07-05T04:14:49.935555+00:00","updated_at":"2026-07-05T04:14:49.935555+00:00"}