{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7UNGNVKSTXADLJWFMXPKYVXKQA","short_pith_number":"pith:7UNGNVKS","schema_version":"1.0","canonical_sha256":"fd1a66d5529dc035a6c565deac56ea8007f11102690eddd95e80c3c9ce37eca2","source":{"kind":"arxiv","id":"2509.00215","version":2},"attestation_state":"computed","paper":{"title":"First Order Model-Based RL through Decoupled Backpropagation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Elliot Chane-Sane, Joseph Amigo, Ludovic Righetti, Nicolas Mansard, Rooholla Khorrambakht","submitted_at":"2025-08-29T19:55:25Z","abstract_excerpt":"There is growing interest in reinforcement learning (RL) methods that leverage the simulator's derivatives to improve learning efficiency. While early gradient-based approaches have demonstrated superior performance compared to derivative-free methods, accessing simulator gradients is often impractical due to their implementation cost or unavailability. Model-based RL (MBRL) can approximate these gradients via learned dynamics models, but the solver efficiency suffers from compounding prediction errors during training rollouts, which can degrade policy performance. We propose an approach that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.00215","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2025-08-29T19:55:25Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"1e41ef6912b403dc5350831f86437a5409beff63c87921f53704bf32a79bb11b","abstract_canon_sha256":"f2fec9a588fc3b25590ac70a195e229655445ed927c402d365c4aec8ea8a65ce"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:40.420762Z","signature_b64":"bI8okOuGiDgPkETWkcEr72dpKhMjNV+nV+8wOIASU7y0f+t6RTews+wiul6C677wbTgNxe3tuZ3wGWMlknJFAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd1a66d5529dc035a6c565deac56ea8007f11102690eddd95e80c3c9ce37eca2","last_reissued_at":"2026-07-05T12:04:40.420001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:40.420001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"First Order Model-Based RL through Decoupled Backpropagation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Elliot Chane-Sane, Joseph Amigo, Ludovic Righetti, Nicolas Mansard, Rooholla Khorrambakht","submitted_at":"2025-08-29T19:55:25Z","abstract_excerpt":"There is growing interest in reinforcement learning (RL) methods that leverage the simulator's derivatives to improve learning efficiency. While early gradient-based approaches have demonstrated superior performance compared to derivative-free methods, accessing simulator gradients is often impractical due to their implementation cost or unavailability. Model-based RL (MBRL) can approximate these gradients via learned dynamics models, but the solver efficiency suffers from compounding prediction errors during training rollouts, which can degrade policy performance. We propose an approach that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.00215","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.00215/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.00215","created_at":"2026-07-05T12:04:40.420116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.00215v2","created_at":"2026-07-05T12:04:40.420116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.00215","created_at":"2026-07-05T12:04:40.420116+00:00"},{"alias_kind":"pith_short_12","alias_value":"7UNGNVKSTXAD","created_at":"2026-07-05T12:04:40.420116+00:00"},{"alias_kind":"pith_short_16","alias_value":"7UNGNVKSTXADLJWF","created_at":"2026-07-05T12:04:40.420116+00:00"},{"alias_kind":"pith_short_8","alias_value":"7UNGNVKS","created_at":"2026-07-05T12:04:40.420116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA","json":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA.json","graph_json":"https://pith.science/api/pith-number/7UNGNVKSTXADLJWFMXPKYVXKQA/graph.json","events_json":"https://pith.science/api/pith-number/7UNGNVKSTXADLJWFMXPKYVXKQA/events.json","paper":"https://pith.science/paper/7UNGNVKS"},"agent_actions":{"view_html":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA","download_json":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA.json","view_paper":"https://pith.science/paper/7UNGNVKS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.00215&json=true","fetch_graph":"https://pith.science/api/pith-number/7UNGNVKSTXADLJWFMXPKYVXKQA/graph.json","fetch_events":"https://pith.science/api/pith-number/7UNGNVKSTXADLJWFMXPKYVXKQA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA/action/storage_attestation","attest_author":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA/action/author_attestation","sign_citation":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA/action/citation_signature","submit_replication":"https://pith.science/pith/7UNGNVKSTXADLJWFMXPKYVXKQA/action/replication_record"}},"created_at":"2026-07-05T12:04:40.420116+00:00","updated_at":"2026-07-05T12:04:40.420116+00:00"}