{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JEUTA4UFE6UMTTHOMCP2R5K6I6","short_pith_number":"pith:JEUTA4UF","schema_version":"1.0","canonical_sha256":"492930728527a8c9ccee609fa8f55e4794662675a556f6f2769174d1e7ca11dc","source":{"kind":"arxiv","id":"2412.12089","version":2},"attestation_state":"computed","paper":{"title":"Stabilizing Reinforcement Learning in Differentiable Multiphysics Simulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Eliot Xing, Jean Oh, Vernon Luk","submitted_at":"2024-12-16T18:56:24Z","abstract_excerpt":"Recent advances in GPU-based parallel simulation have enabled practitioners to collect large amounts of data and train complex control policies using deep reinforcement learning (RL), on commodity GPUs. However, such successes for RL in robotics have been limited to tasks sufficiently simulated by fast rigid-body dynamics. Simulation techniques for soft bodies are comparatively several orders of magnitude slower, thereby limiting the use of RL due to sample complexity requirements. To address this challenge, this paper presents both a novel RL algorithm and a simulation platform to enable scal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.12089","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-16T18:56:24Z","cross_cats_sorted":["cs.AI","cs.CV","cs.RO"],"title_canon_sha256":"b1affe635336396ed6a3b5d987d2c1a0aa3660b7ecd62299b8e98b5a53dea5c1","abstract_canon_sha256":"85ffc33cf306bdabf5354516a27f8d06a22b508bc3f64b40bb2e274a9896f0ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:27.332964Z","signature_b64":"+HMisRVMaw8KfA/pU3EBbtT1utdRyCvcXJcy6h2uqCukXRimQI6VODxp1u5Ob/jhjuvxM6ddCss0/i8NEptvBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"492930728527a8c9ccee609fa8f55e4794662675a556f6f2769174d1e7ca11dc","last_reissued_at":"2026-07-05T10:21:27.332398Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:27.332398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Stabilizing Reinforcement Learning in Differentiable Multiphysics Simulation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Eliot Xing, Jean Oh, Vernon Luk","submitted_at":"2024-12-16T18:56:24Z","abstract_excerpt":"Recent advances in GPU-based parallel simulation have enabled practitioners to collect large amounts of data and train complex control policies using deep reinforcement learning (RL), on commodity GPUs. However, such successes for RL in robotics have been limited to tasks sufficiently simulated by fast rigid-body dynamics. Simulation techniques for soft bodies are comparatively several orders of magnitude slower, thereby limiting the use of RL due to sample complexity requirements. To address this challenge, this paper presents both a novel RL algorithm and a simulation platform to enable scal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.12089","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.12089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.12089","created_at":"2026-07-05T10:21:27.332462+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.12089v2","created_at":"2026-07-05T10:21:27.332462+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.12089","created_at":"2026-07-05T10:21:27.332462+00:00"},{"alias_kind":"pith_short_12","alias_value":"JEUTA4UFE6UM","created_at":"2026-07-05T10:21:27.332462+00:00"},{"alias_kind":"pith_short_16","alias_value":"JEUTA4UFE6UMTTHO","created_at":"2026-07-05T10:21:27.332462+00:00"},{"alias_kind":"pith_short_8","alias_value":"JEUTA4UF","created_at":"2026-07-05T10:21:27.332462+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18161","citing_title":"Does \"Do Differentiable Simulators Give Better Policy Gradients?'' Give Better Policy Gradients?","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6","json":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6.json","graph_json":"https://pith.science/api/pith-number/JEUTA4UFE6UMTTHOMCP2R5K6I6/graph.json","events_json":"https://pith.science/api/pith-number/JEUTA4UFE6UMTTHOMCP2R5K6I6/events.json","paper":"https://pith.science/paper/JEUTA4UF"},"agent_actions":{"view_html":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6","download_json":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6.json","view_paper":"https://pith.science/paper/JEUTA4UF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.12089&json=true","fetch_graph":"https://pith.science/api/pith-number/JEUTA4UFE6UMTTHOMCP2R5K6I6/graph.json","fetch_events":"https://pith.science/api/pith-number/JEUTA4UFE6UMTTHOMCP2R5K6I6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6/action/storage_attestation","attest_author":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6/action/author_attestation","sign_citation":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6/action/citation_signature","submit_replication":"https://pith.science/pith/JEUTA4UFE6UMTTHOMCP2R5K6I6/action/replication_record"}},"created_at":"2026-07-05T10:21:27.332462+00:00","updated_at":"2026-07-05T10:21:27.332462+00:00"}